diff --git a/AUTHORS b/AUTHORS
index a0fbe722f..1392aa71f 100644
--- a/AUTHORS
+++ b/AUTHORS
@@ -1,28 +1,25 @@
Scrapy was brought to life by Shane Evans while hacking a scraping framework
prototype for Mydeco (mydeco.com). It soon became maintained, extended and
-improved by Insophia (insophia.com), with the sponsorship of By Design (the
-company behind Mydeco).
+improved by Insophia (insophia.com), with the initial sponsorship of Mydeco to
+bootstrap the project.
-Here is the list of the primary authors & contributors, along with their user
-name (in Scrapy trac/subversion). Emails are intentionally left out to avoid
-spam.
+Here is the list of the primary authors & contributors:
- * Pablo Hoffman (pablo)
- * Daniel Graña (daniel)
- * Martin Olveyra (olveyra)
- * Gabriel García (elpolilla)
- * Michael Cetrulo (samus_)
- * Artem Bogomyagkov (artem)
- * Damian Canabal (calarval)
- * Andres Moreira (andres)
- * Ismael Carnales (ismael)
- * Matías Aguirre (omab)
- * German Hoffman (german)
- * Anibal Pacheco (anibal)
+ * Pablo Hoffman
+ * Daniel Graña
+ * Martin Olveyra
+ * Gabriel García
+ * Michael Cetrulo
+ * Artem Bogomyagkov
+ * Damian Canabal
+ * Andres Moreira
+ * Ismael Carnales
+ * Matías Aguirre
+ * German Hoffmann
+ * Anibal Pacheco
* Bruno Deferrari
* Shane Evans
-
-And here is the list of people who have helped to put the Scrapy homepage live:
-
- * Ezequiel Rivero (ezequiel)
+ * Ezequiel Rivero
+ * Patrick Mezard
+ * Rolando Espinoza
diff --git a/docs/experimental/crawlspider-v2.rst b/docs/experimental/crawlspider-v2.rst
new file mode 100644
index 000000000..a1ea09c48
--- /dev/null
+++ b/docs/experimental/crawlspider-v2.rst
@@ -0,0 +1,128 @@
+.. _topics-crawlspider-v2:
+
+==============
+CrawlSpider v2
+==============
+
+Introduction
+============
+
+TODO: introduction
+
+Rules Matching
+==============
+
+TODO: describe purpose of rules
+
+Request Extractors & Processors
+===============================
+
+TODO: describe purpose of extractors & processors
+
+Examples
+========
+
+TODO: plenty of examples
+
+
+.. module:: scrapy.contrib_exp.crawlspider.spider
+ :synopsis: CrawlSpider
+
+
+Reference
+=========
+
+CrawlSpider
+-----------
+
+TODO: describe crawlspider
+
+.. class:: CrawlSpider
+
+ TODO: describe class
+
+
+.. module:: scrapy.contrib_exp.crawlspider.rules
+ :synopsis: Rules
+
+Rules
+-----
+
+TODO: describe spider rules
+
+.. class:: Rule
+
+ TODO: describe Rules class
+
+
+.. module:: scrapy.contrib_exp.crawlspider.reqext
+ :synopsis: Request Extractors
+
+Request Extractors
+------------------
+
+TODO: describe extractors purpose
+
+.. class:: BaseSgmlRequestExtractor
+
+ TODO: describe base extractor
+
+.. class:: SgmlRequestExtractor
+
+ TODO: describe sgml extractor
+
+.. class:: XPathRequestExtractor
+
+ TODO: describe xpath request extractor
+
+
+.. module:: scrapy.contrib_exp.crawlspider.reqproc
+ :synopsis: Request Processors
+
+Request Processors
+------------------
+
+TODO: describe request processors
+
+.. class:: Canonicalize
+
+ TODO: describe proc
+
+.. class:: Unique
+
+ TODO: describe unique
+
+.. class:: FilterDomain
+
+ TODO: describe filter domain
+
+.. class:: FilterUrl
+
+ TODO: describe filter url
+
+
+.. module:: scrapy.contrib_exp.crawlspider.matchers
+ :synopsis: Matchers
+
+Request/Response Matchers
+-------------------------
+
+TODO: describe matchers
+
+.. class:: BaseMatcher
+
+ TODO: describe base matcher
+
+.. class:: UrlMatcher
+
+ TODO: describe url matcher
+
+.. class:: UrlRegexMatcher
+
+ TODO: describe UrlListMatcher
+
+.. class:: UrlListMatcher
+
+ TODO: describe url list matcher
+
+
diff --git a/docs/experimental/index.rst b/docs/experimental/index.rst
index 47f4ee4ac..f63ceddd9 100644
--- a/docs/experimental/index.rst
+++ b/docs/experimental/index.rst
@@ -21,3 +21,4 @@ it's properly merged) . Use at your own risk.
djangoitems
scheduler-middleware
+ crawlspider-v2
diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst
index 3fb965110..6d7c3c64a 100644
--- a/docs/intro/tutorial.rst
+++ b/docs/intro/tutorial.rst
@@ -420,8 +420,8 @@ scraped so far, the code for our Spider should be like this::
Now doing a crawl on the dmoz.org domain yields ``DmozItem``'s::
- [dmoz.org] DEBUG: Scraped DmozItem({'title': [u'Text Processing in Python'], 'link': [u'http://gnosis.cx/TPiP/'], 'desc': [u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n']}) in
- [dmoz.org] DEBUG: Scraped DmozItem({'title': [u'XML Processing with Python'], 'link': [u'http://www.informit.com/store/product.aspx?isbn=0130211192'], 'desc': [u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n']}) in
+ [dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n'], link=[u'http://gnosis.cx/TPiP/'], title=[u'Text Processing in Python']) in
+ [dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n'], link=[u'http://www.informit.com/store/product.aspx?isbn=0130211192'], title=[u'XML Processing with Python']) in
Storing the data (using an Item Pipeline)
diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst
index a72b285c3..019c9c32b 100644
--- a/docs/topics/item-pipeline.rst
+++ b/docs/topics/item-pipeline.rst
@@ -98,10 +98,10 @@ spider returns multiples items with the same id::
del self.duplicates[spider]
def process_item(self, spider, item):
- if item.id in self.duplicates[spider]:
+ if item['id'] in self.duplicates[spider]:
raise DropItem("Duplicate item found: %s" % item)
else:
- self.duplicates[spider].add(item.id)
+ self.duplicates[spider].add(item['id'])
return item
Built-in Item Pipelines reference
@@ -178,7 +178,7 @@ on the respective Item Exporter to get more info.
the first line. This format requires you to specify the fields to export
using the :setting:`EXPORT_FIELDS` setting.
-* ``jsonlines``: uses a :class:`~jsonlines.JsonLinesItemExporter`
+* ``json``: uses a :class:`~jsonlines.JsonLinesItemExporter`
* ``pickle``: uses a :class:`PickleItemExporter`
diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst
index 3472b1937..9f4dcdc46 100644
--- a/docs/topics/request-response.rst
+++ b/docs/topics/request-response.rst
@@ -466,12 +466,14 @@ TextResponse objects
.. attribute:: TextResponse.encoding
- A string with the encoding of this response. The encoding is resolved in the
- following order:
+ A string with the encoding of this response. The encoding is resolved by
+ trying the following mechanisms, in order:
1. the encoding passed in the constructor `encoding` argument
- 2. the encoding declared in the Content-Type HTTP header
+ 2. the encoding declared in the Content-Type HTTP header. If this
+ encoding is not valid (ie. unknown), it is ignored and the next
+ resolution mechanism is tried.
3. the encoding declared in the response body. The TextResponse class
doesn't provide any special functionality for this. However, the
diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst
index 4b3dbc78f..4058eec3e 100644
--- a/docs/topics/settings.rst
+++ b/docs/topics/settings.rst
@@ -418,6 +418,15 @@ supported. Example::
DOWNLOAD_DELAY = 0.25 # 250 ms of delay
+This setting is also affected by the :setting:`RANDOMIZE_DOWNLOAD_DELAY`
+setting (which is enabled by default). By default, Scrapy doesn't wait a fixed
+amount of time between requests, but uses a random interval between 0.5 and 1.5
+* :setting:`DOWNLOAD_DELAY`.
+
+Another way to change the download delay (per spider, instead of globally) is
+by using the ``download_delay`` spider attribute, which takes more precedence
+than this setting.
+
.. setting:: DOWNLOAD_TIMEOUT
DOWNLOAD_TIMEOUT
@@ -677,6 +686,27 @@ Example::
NEWSPIDER_MODULE = 'mybot.spiders_dev'
+.. setting:: RANDOMIZE_DOWNLOAD_DELAY
+
+RANDOMIZE_DOWNLOAD_DELAY
+------------------------
+
+Default: ``True``
+
+If enabled, Scrapy will wait a random amount of time (between 0.5 and 1.5
+* :setting:`DOWNLOAD_DELAY`) while fetching requests from the same
+spider.
+
+This randomization decreases the chance of the crawler being detected (and
+subsequently blocked) by sites which analyze requests looking for statistically
+significant similarities in the time between their times.
+
+The randomization policy is the same used by `wget`_ ``--random-wait`` option.
+
+If :setting:`DOWNLOAD_DELAY` is zero (default) this option has no effect.
+
+.. _wget: http://www.gnu.org/software/wget/manual/wget.html
+
.. setting:: REDIRECT_MAX_TIMES
REDIRECT_MAX_TIMES
@@ -773,7 +803,7 @@ The scheduler to use for crawling.
SCHEDULER_ORDER
---------------
-Default: ``'BFO'``
+Default: ``'DFO'``
Scope: ``scrapy.core.scheduler``
diff --git a/examples/experimental/googledir/googledir/__init__.py b/examples/experimental/googledir/googledir/__init__.py
new file mode 100644
index 000000000..3104ef709
--- /dev/null
+++ b/examples/experimental/googledir/googledir/__init__.py
@@ -0,0 +1 @@
+# googledir project
diff --git a/examples/experimental/googledir/googledir/items.py b/examples/experimental/googledir/googledir/items.py
new file mode 100644
index 000000000..decc2c9ba
--- /dev/null
+++ b/examples/experimental/googledir/googledir/items.py
@@ -0,0 +1,16 @@
+# Define here the models for your scraped items
+#
+# See documentation in:
+# http://doc.scrapy.org/topics/items.html
+
+from scrapy.item import Item, Field
+
+class GoogledirItem(Item):
+
+ name = Field(default='')
+ url = Field(default='')
+ description = Field(default='')
+
+ def __str__(self):
+ return "Google Category: name=%s url=%s" \
+ % (self['name'], self['url'])
diff --git a/examples/experimental/googledir/googledir/pipelines.py b/examples/experimental/googledir/googledir/pipelines.py
new file mode 100644
index 000000000..f775b254c
--- /dev/null
+++ b/examples/experimental/googledir/googledir/pipelines.py
@@ -0,0 +1,22 @@
+# Define your item pipelines here
+#
+# Don't forget to add your pipeline to the ITEM_PIPELINES setting
+# See: http://doc.scrapy.org/topics/item-pipeline.html
+
+from scrapy.core.exceptions import DropItem
+
+class FilterWordsPipeline(object):
+ """
+ A pipeline for filtering out items which contain certain
+ words in their description
+ """
+
+ # put all words in lowercase
+ words_to_filter = ['politics', 'religion']
+
+ def process_item(self, spider, item):
+ for word in self.words_to_filter:
+ if word in unicode(item['description']).lower():
+ raise DropItem("Contains forbidden word: %s" % word)
+ else:
+ return item
diff --git a/examples/experimental/googledir/googledir/settings.py b/examples/experimental/googledir/googledir/settings.py
new file mode 100644
index 000000000..4e3c11163
--- /dev/null
+++ b/examples/experimental/googledir/googledir/settings.py
@@ -0,0 +1,21 @@
+# Scrapy settings for googledir project
+#
+# For simplicity, this file contains only the most important settings by
+# default. All the other settings are documented here:
+#
+# http://doc.scrapy.org/topics/settings.html
+#
+# Or you can copy and paste them from where they're defined in Scrapy:
+#
+# scrapy/conf/default_settings.py
+#
+
+BOT_NAME = 'googledir'
+BOT_VERSION = '1.0'
+
+SPIDER_MODULES = ['googledir.spiders']
+NEWSPIDER_MODULE = 'googledir.spiders'
+DEFAULT_ITEM_CLASS = 'googledir.items.GoogledirItem'
+USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION)
+
+ITEM_PIPELINES = ['googledir.pipelines.FilterWordsPipeline']
diff --git a/examples/experimental/googledir/googledir/spiders/__init__.py b/examples/experimental/googledir/googledir/spiders/__init__.py
new file mode 100644
index 000000000..5065ccba5
--- /dev/null
+++ b/examples/experimental/googledir/googledir/spiders/__init__.py
@@ -0,0 +1,8 @@
+# This package will contain the spiders of your Scrapy project
+#
+# To create the first spider for your project use this command:
+#
+# scrapy-ctl.py genspider myspider myspider-domain.com
+#
+# For more info see:
+# http://doc.scrapy.org/topics/spiders.html
diff --git a/examples/experimental/googledir/googledir/spiders/google_directory.py b/examples/experimental/googledir/googledir/spiders/google_directory.py
new file mode 100644
index 000000000..fdbf3f24a
--- /dev/null
+++ b/examples/experimental/googledir/googledir/spiders/google_directory.py
@@ -0,0 +1,40 @@
+from scrapy.selector import HtmlXPathSelector
+from scrapy.contrib.loader import XPathItemLoader
+from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule
+
+from googledir.items import GoogledirItem
+
+class GoogleDirectorySpider(CrawlSpider):
+
+ domain_name = 'directory.google.com'
+ start_urls = ['http://directory.google.com/']
+
+ rules = (
+ # search for categories pattern and follow links
+ Rule(r'/[A-Z][a-zA-Z_/]+$', 'parse_category', follow=True),
+ )
+
+ def parse_category(self, response):
+ # The main selector we're using to extract data from the page
+ main_selector = HtmlXPathSelector(response)
+
+ # The XPath to website links in the directory page
+ xpath = '//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td/font'
+
+ # Get a list of (sub) selectors to each website node pointed by the XPath
+ sub_selectors = main_selector.select(xpath)
+
+ # Iterate over the sub-selectors to extract data for each website
+ for selector in sub_selectors:
+ item = GoogledirItem()
+
+ l = XPathItemLoader(item=item, selector=selector)
+ l.add_xpath('name', 'a/text()')
+ l.add_xpath('url', 'a/@href')
+ l.add_xpath('description', 'font[2]/text()')
+
+ # Here we populate the item and yield it
+ yield l.load_item()
+
+SPIDER = GoogleDirectorySpider()
+
diff --git a/examples/experimental/googledir/scrapy-ctl.py b/examples/experimental/googledir/scrapy-ctl.py
new file mode 100644
index 000000000..552421ac3
--- /dev/null
+++ b/examples/experimental/googledir/scrapy-ctl.py
@@ -0,0 +1,7 @@
+#!/usr/bin/env python
+
+import os
+os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'googledir.settings')
+
+from scrapy.command.cmdline import execute
+execute()
diff --git a/examples/experimental/imdb/imdb/__init__.py b/examples/experimental/imdb/imdb/__init__.py
new file mode 100644
index 000000000..5bb534f79
--- /dev/null
+++ b/examples/experimental/imdb/imdb/__init__.py
@@ -0,0 +1 @@
+# package
diff --git a/examples/experimental/imdb/imdb/items.py b/examples/experimental/imdb/imdb/items.py
new file mode 100644
index 000000000..03bb5c2c3
--- /dev/null
+++ b/examples/experimental/imdb/imdb/items.py
@@ -0,0 +1,12 @@
+# Define here the models for your scraped items
+#
+# See documentation in:
+# http://doc.scrapy.org/topics/items.html
+
+from scrapy.item import Item, Field
+
+class ImdbItem(Item):
+ # define the fields for your item here like:
+ # name = Field()
+ title = Field()
+ url = Field()
diff --git a/examples/experimental/imdb/imdb/pipelines.py b/examples/experimental/imdb/imdb/pipelines.py
new file mode 100644
index 000000000..e60714159
--- /dev/null
+++ b/examples/experimental/imdb/imdb/pipelines.py
@@ -0,0 +1,8 @@
+# Define your item pipelines here
+#
+# Don't forget to add your pipeline to the ITEM_PIPELINES setting
+# See: http://doc.scrapy.org/topics/item-pipeline.html
+
+class ImdbPipeline(object):
+ def process_item(self, spider, item):
+ return item
diff --git a/examples/experimental/imdb/imdb/settings.py b/examples/experimental/imdb/imdb/settings.py
new file mode 100644
index 000000000..de026dc14
--- /dev/null
+++ b/examples/experimental/imdb/imdb/settings.py
@@ -0,0 +1,20 @@
+# Scrapy settings for imdb project
+#
+# For simplicity, this file contains only the most important settings by
+# default. All the other settings are documented here:
+#
+# http://doc.scrapy.org/topics/settings.html
+#
+# Or you can copy and paste them from where they're defined in Scrapy:
+#
+# scrapy/conf/default_settings.py
+#
+
+BOT_NAME = 'imdb'
+BOT_VERSION = '1.0'
+
+SPIDER_MODULES = ['imdb.spiders']
+NEWSPIDER_MODULE = 'imdb.spiders'
+DEFAULT_ITEM_CLASS = 'imdb.items.ImdbItem'
+USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION)
+
diff --git a/examples/experimental/imdb/imdb/spiders/__init__.py b/examples/experimental/imdb/imdb/spiders/__init__.py
new file mode 100644
index 000000000..5065ccba5
--- /dev/null
+++ b/examples/experimental/imdb/imdb/spiders/__init__.py
@@ -0,0 +1,8 @@
+# This package will contain the spiders of your Scrapy project
+#
+# To create the first spider for your project use this command:
+#
+# scrapy-ctl.py genspider myspider myspider-domain.com
+#
+# For more info see:
+# http://doc.scrapy.org/topics/spiders.html
diff --git a/examples/experimental/imdb/imdb/spiders/imdb_site.py b/examples/experimental/imdb/imdb/spiders/imdb_site.py
new file mode 100644
index 000000000..5c3f3c06b
--- /dev/null
+++ b/examples/experimental/imdb/imdb/spiders/imdb_site.py
@@ -0,0 +1,140 @@
+from scrapy.http import Request
+from scrapy.selector import HtmlXPathSelector
+from scrapy.contrib.loader import XPathItemLoader
+from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize, \
+ FilterDupes, FilterUrl
+from scrapy.utils.url import urljoin_rfc
+
+from imdb.items import ImdbItem, Field
+
+from itertools import chain, imap, izip
+
+class UsaOpeningWeekMovie(ImdbItem):
+ pass
+
+class UsaTopWeekMovie(ImdbItem):
+ pass
+
+class Top250Movie(ImdbItem):
+ rank = Field()
+ rating = Field()
+ year = Field()
+ votes = Field()
+
+class MovieItem(ImdbItem):
+ release_date = Field()
+ tagline = Field()
+
+
+class ImdbSiteSpider(CrawlSpider):
+ domain_name = 'imdb.com'
+ start_urls = ['http://www.imdb.com/']
+
+ # extract requests using this classes from urls matching 'follow' flag
+ request_extractors = [
+ SgmlRequestExtractor(tags=['a'], attrs=['href']),
+ ]
+
+ # process requests using this classes from urls matching 'follow' flag
+ request_processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterUrl(deny=r'/tt\d+/$'), # deny movie url as we will dispatch
+ # manually the movie requests
+ ]
+
+ # include domain bit for demo purposes
+ rules = (
+ # these two rules expects requests from start url
+ Rule(r'imdb.com/nowplaying/$', 'parse_now_playing'),
+ Rule(r'imdb.com/chart/top$', 'parse_top_250'),
+ # this rule will parse requests manually dispatched
+ Rule(r'imdb.com/title/tt\d+/$', 'parse_movie_info'),
+ )
+
+ def parse_now_playing(self, response):
+ """Scrapes USA openings this week and top 10 in week"""
+ self.log("Parsing USA Top Week")
+ hxs = HtmlXPathSelector(response)
+
+ _urljoin = lambda url: self._urljoin(response, url)
+
+ #
+ # openings this week
+ #
+ openings = hxs.select('//table[@class="movies"]//a[@class="title"]')
+ boxoffice = hxs.select('//table[@class="boxoffice movies"]//a[@class="title"]')
+
+ opening_titles = openings.select('text()').extract()
+ opening_urls = imap(_urljoin, openings.select('@href').extract())
+
+ box_titles = boxoffice.select('text()').extract()
+ box_urls = imap(_urljoin, boxoffice.select('@href').extract())
+
+ # items
+ opening_items = (UsaOpeningWeekMovie(title=title, url=url)
+ for (title, url)
+ in izip(opening_titles, opening_urls))
+
+ box_items = (UsaTopWeekMovie(title=title, url=url)
+ for (title, url)
+ in izip(box_titles, box_urls))
+
+ # movie requests
+ requests = imap(self.make_requests_from_url,
+ chain(opening_urls, box_urls))
+
+ return chain(opening_items, box_items, requests)
+
+ def parse_top_250(self, response):
+ """Scrapes movies from top 250 list"""
+ self.log("Parsing Top 250")
+ hxs = HtmlXPathSelector(response)
+
+ # scrap each row in the table
+ rows = hxs.select('//div[@id="main"]/table/tr//a/ancestor::tr')
+ for row in rows:
+ fields = row.select('td//text()').extract()
+ url, = row.select('td//a/@href').extract()
+ url = self._urljoin(response, url)
+
+ item = Top250Movie()
+ item['title'] = fields[2]
+ item['url'] = url
+ item['rank'] = fields[0]
+ item['rating'] = fields[1]
+ item['year'] = fields[3]
+ item['votes'] = fields[4]
+
+ # scrapped top250 item
+ yield item
+ # fetch movie
+ yield self.make_requests_from_url(url)
+
+ def parse_movie_info(self, response):
+ """Scrapes movie information"""
+ self.log("Parsing Movie Info")
+ hxs = HtmlXPathSelector(response)
+ selector = hxs.select('//div[@class="maindetails"]')
+
+ item = MovieItem()
+ # set url
+ item['url'] = response.url
+
+ # use item loader for other attributes
+ l = XPathItemLoader(item=item, selector=selector)
+ l.add_xpath('title', './/h1/text()')
+ l.add_xpath('release_date', './/h5[text()="Release Date:"]'
+ '/following-sibling::div/text()')
+ l.add_xpath('tagline', './/h5[text()="Tagline:"]'
+ '/following-sibling::div/text()')
+
+ yield l.load_item()
+
+ def _urljoin(self, response, url):
+ """Helper to convert relative urls to absolute"""
+ return urljoin_rfc(response.url, url, response.encoding)
+
+SPIDER = ImdbSiteSpider()
diff --git a/examples/experimental/imdb/scrapy-ctl.py b/examples/experimental/imdb/scrapy-ctl.py
new file mode 100644
index 000000000..df57621b3
--- /dev/null
+++ b/examples/experimental/imdb/scrapy-ctl.py
@@ -0,0 +1,7 @@
+#!/usr/bin/env python
+
+import os
+os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'imdb.settings')
+
+from scrapy.command.cmdline import execute
+execute()
diff --git a/scrapy/__init__.py b/scrapy/__init__.py
index 0051aa75e..6f8ca15e6 100644
--- a/scrapy/__init__.py
+++ b/scrapy/__init__.py
@@ -2,8 +2,8 @@
Scrapy - a screen scraping framework written in Python
"""
-version_info = (0, 8, 0, '', 0)
-__version__ = "0.8"
+version_info = (0, 9, 0, 'dev')
+__version__ = "0.9-dev"
import sys, os
diff --git a/scrapy/command/cmdline.py b/scrapy/command/cmdline.py
index 25f531cea..1a190549d 100644
--- a/scrapy/command/cmdline.py
+++ b/scrapy/command/cmdline.py
@@ -7,20 +7,14 @@ import cProfile
import scrapy
from scrapy import log
-from scrapy.spider import spiders
from scrapy.xlib import lsprofcalltree
from scrapy.conf import settings
from scrapy.command.models import ScrapyCommand
+from scrapy.utils.signal import send_catch_log
-# This dict holds information about the executed command for later use
-command_executed = {}
-
-def _save_command_executed(cmdname, cmd, args, opts):
- """Save command executed info for later reference"""
- command_executed['name'] = cmdname
- command_executed['class'] = cmd
- command_executed['args'] = args[:]
- command_executed['opts'] = opts.__dict__.copy()
+# Signal that carries information about the command which was executed
+# args: cmdname, cmdobj, args, opts
+command_executed = object()
def _find_commands(dir):
try:
@@ -127,7 +121,8 @@ def execute(argv=None):
sys.exit(2)
del args[0] # remove command name from args
- _save_command_executed(cmdname, cmd, args, opts)
+ send_catch_log(signal=command_executed, cmdname=cmdname, cmdobj=cmd, \
+ args=args, opts=opts)
from scrapy.core.manager import scrapymanager
scrapymanager.configure(control_reactor=True)
ret = _run_command(cmd, args, opts)
@@ -136,23 +131,25 @@ def execute(argv=None):
def _run_command(cmd, args, opts):
if opts.profile or opts.lsprof:
- if opts.profile:
- log.msg("writing cProfile stats to %r" % opts.profile)
- if opts.lsprof:
- log.msg("writing lsprof stats to %r" % opts.lsprof)
- loc = locals()
- p = cProfile.Profile()
- p.runctx('ret = cmd.run(args, opts)', globals(), loc)
- if opts.profile:
- p.dump_stats(opts.profile)
- k = lsprofcalltree.KCacheGrind(p)
- if opts.lsprof:
- with open(opts.lsprof, 'w') as f:
- k.output(f)
- ret = loc['ret']
+ return _run_command_profiled(cmd, args, opts)
else:
- ret = cmd.run(args, opts)
- return ret
+ return cmd.run(args, opts)
+
+def _run_command_profiled(cmd, args, opts):
+ if opts.profile:
+ log.msg("writing cProfile stats to %r" % opts.profile)
+ if opts.lsprof:
+ log.msg("writing lsprof stats to %r" % opts.lsprof)
+ loc = locals()
+ p = cProfile.Profile()
+ p.runctx('ret = cmd.run(args, opts)', globals(), loc)
+ if opts.profile:
+ p.dump_stats(opts.profile)
+ k = lsprofcalltree.KCacheGrind(p)
+ if opts.lsprof:
+ with open(opts.lsprof, 'w') as f:
+ k.output(f)
+ return loc['ret']
if __name__ == '__main__':
execute()
diff --git a/scrapy/conf/default_settings.py b/scrapy/conf/default_settings.py
index 8892e41b4..3ee377a44 100644
--- a/scrapy/conf/default_settings.py
+++ b/scrapy/conf/default_settings.py
@@ -122,6 +122,8 @@ MYSQL_CONNECTION_SETTINGS = {}
NEWSPIDER_MODULE = ''
+RANDOMIZE_DOWNLOAD_DELAY = True
+
REDIRECT_MAX_METAREFRESH_DELAY = 100
REDIRECT_MAX_TIMES = 20 # uses Firefox default setting
REDIRECT_PRIORITY_ADJUST = +2
@@ -150,7 +152,7 @@ SCHEDULER_MIDDLEWARES_BASE = {
'scrapy.contrib.schedulermiddleware.duplicatesfilter.DuplicatesFilterMiddleware': 500,
}
-SCHEDULER_ORDER = 'BFO' # available orders: BFO (default), DFO
+SCHEDULER_ORDER = 'DFO'
SPIDER_MANAGER_CLASS = 'scrapy.contrib.spidermanager.TwistedPluginSpiderManager'
diff --git a/scrapy/contrib/groupsettings.py b/scrapy/contrib/groupsettings.py
deleted file mode 100644
index 4bad4100b..000000000
--- a/scrapy/contrib/groupsettings.py
+++ /dev/null
@@ -1,26 +0,0 @@
-"""
-Extensions to override scrapy settings with per-group settings according to the
-group the spider belongs to. It only overrides the settings when running the
-crawl command with *only one domain as argument*.
-"""
-
-from scrapy.conf import settings
-from scrapy.core.exceptions import NotConfigured
-from scrapy.command.cmdline import command_executed
-
-class GroupSettings(object):
-
- def __init__(self):
- if not settings.getbool("GROUPSETTINGS_ENABLED"):
- raise NotConfigured
-
- if command_executed and command_executed['name'] == 'crawl':
- mod = __import__(settings['GROUPSETTINGS_MODULE'], {}, {}, [''])
- args = command_executed['args']
- if len(args) == 1 and not args[0].startswith('http://'):
- domain = args[0]
- settings.overrides.update(mod.default_settings)
- for group, domains in mod.group_spiders.iteritems():
- if domain in domains:
- settings.overrides.update(mod.group_settings.get(group, {}))
-
diff --git a/scrapy/contrib_exp/crawlspider/__init__.py b/scrapy/contrib_exp/crawlspider/__init__.py
new file mode 100644
index 000000000..03173eb38
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/__init__.py
@@ -0,0 +1,4 @@
+"""CrawlSpider v2"""
+
+from .rules import Rule
+from .spider import CrawlSpider
diff --git a/scrapy/contrib_exp/crawlspider/matchers.py b/scrapy/contrib_exp/crawlspider/matchers.py
new file mode 100644
index 000000000..3ef259c67
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/matchers.py
@@ -0,0 +1,61 @@
+"""
+Request/Response Matchers
+
+Perform evaluation to Request or Response attributes
+"""
+
+import re
+
+class BaseMatcher(object):
+ """Base matcher. Returns True by default."""
+
+ def matches_request(self, request):
+ """Performs Request Matching"""
+ return True
+
+ def matches_response(self, response):
+ """Performs Response Matching"""
+ return True
+
+
+class UrlMatcher(BaseMatcher):
+ """Matches URL attribute"""
+
+ def __init__(self, url):
+ """Initialize url attribute"""
+ self._url = url
+
+ def matches_url(self, url):
+ """Returns True if given url is equal to matcher's url"""
+ return self._url == url
+
+ def matches_request(self, request):
+ """Returns True if Request's url matches initial url"""
+ return self.matches_url(request.url)
+
+ def matches_response(self, response):
+ """Returns True if Response's url matches initial url"""
+ return self.matches_url(response.url)
+
+
+class UrlRegexMatcher(UrlMatcher):
+ """Matches URL using regular expression"""
+
+ def __init__(self, regex, flags=0):
+ """Initialize regular expression"""
+ self._regex = re.compile(regex, flags)
+
+ def matches_url(self, url):
+ """Returns True if url matches regular expression"""
+ return self._regex.search(url) is not None
+
+
+class UrlListMatcher(UrlMatcher):
+ """Matches if URL is in List"""
+
+ def __init__(self, urls):
+ self._urls = urls
+
+ def matches_url(self, url):
+ """Returns True if url is in urls list"""
+ return url in self._urls
diff --git a/scrapy/contrib_exp/crawlspider/reqext.py b/scrapy/contrib_exp/crawlspider/reqext.py
new file mode 100644
index 000000000..bb7318f79
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/reqext.py
@@ -0,0 +1,117 @@
+"""Request Extractors"""
+from scrapy.http import Request
+from scrapy.selector import HtmlXPathSelector
+from scrapy.utils.misc import arg_to_iter
+from scrapy.utils.python import FixedSGMLParser, str_to_unicode
+from scrapy.utils.url import safe_url_string, urljoin_rfc
+
+from itertools import ifilter
+
+
+class BaseSgmlRequestExtractor(FixedSGMLParser):
+ """Base SGML Request Extractor"""
+
+ def __init__(self, tag='a', attr='href'):
+ """Initialize attributes"""
+ FixedSGMLParser.__init__(self)
+
+ self.scan_tag = tag if callable(tag) else lambda t: t == tag
+ self.scan_attr = attr if callable(attr) else lambda a: a == attr
+ self.current_request = None
+
+ def extract_requests(self, response):
+ """Returns list of requests extracted from response"""
+ return self._extract_requests(response.body, response.url,
+ response.encoding)
+
+ def _extract_requests(self, response_text, response_url, response_encoding):
+ """Extract requests with absolute urls"""
+ self.reset()
+ self.feed(response_text)
+ self.close()
+
+ base_url = self.base_url if self.base_url else response_url
+ self._make_absolute_urls(base_url, response_encoding)
+ self._fix_link_text_encoding(response_encoding)
+
+ return self.requests
+
+ def _make_absolute_urls(self, base_url, encoding):
+ """Makes all request's urls absolute"""
+ for req in self.requests:
+ url = req.url
+ # make absolute url
+ url = urljoin_rfc(base_url, url, encoding)
+ url = safe_url_string(url, encoding)
+ # replace in-place request's url
+ req.url = url
+
+ def _fix_link_text_encoding(self, encoding):
+ """Convert link_text to unicode for each request"""
+ for req in self.requests:
+ req.meta.setdefault('link_text', '')
+ req.meta['link_text'] = str_to_unicode(req.meta['link_text'],
+ encoding)
+
+ def reset(self):
+ """Reset state"""
+ FixedSGMLParser.reset(self)
+ self.requests = []
+ self.base_url = None
+
+ def unknown_starttag(self, tag, attrs):
+ """Process unknown start tag"""
+ if 'base' == tag:
+ self.base_url = dict(attrs).get('href')
+
+ _matches = lambda (attr, value): self.scan_attr(attr) \
+ and value is not None
+ if self.scan_tag(tag):
+ for attr, value in ifilter(_matches, attrs):
+ req = Request(url=value)
+ self.requests.append(req)
+ self.current_request = req
+
+ def unknown_endtag(self, tag):
+ """Process unknown end tag"""
+ self.current_request = None
+
+ def handle_data(self, data):
+ """Process data"""
+ current = self.current_request
+ if current and not 'link_text' in current.meta:
+ current.meta['link_text'] = data.strip()
+
+
+class SgmlRequestExtractor(BaseSgmlRequestExtractor):
+ """SGML Request Extractor"""
+
+ def __init__(self, tags=None, attrs=None):
+ """Initialize with custom tag & attribute function checkers"""
+ # defaults
+ tags = tuple(tags) if tags else ('a', 'area')
+ attrs = tuple(attrs) if attrs else ('href', )
+
+ tag_func = lambda x: x in tags
+ attr_func = lambda x: x in attrs
+ BaseSgmlRequestExtractor.__init__(self, tag=tag_func, attr=attr_func)
+
+# TODO: move to own file
+class XPathRequestExtractor(SgmlRequestExtractor):
+ """SGML Request Extractor with XPath restriction"""
+
+ def __init__(self, restrict_xpaths, tags=None, attrs=None):
+ """Initialize XPath restrictions"""
+ self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
+ SgmlRequestExtractor.__init__(self, tags, attrs)
+
+ def extract_requests(self, response):
+ """Restrict to XPath regions"""
+ hxs = HtmlXPathSelector(response)
+ fragments = (''.join(
+ html_frag for html_frag in hxs.select(xpath).extract()
+ ) for xpath in self.restrict_xpaths)
+ html_slice = ''.join(html_frag for html_frag in fragments)
+ return self._extract_requests(html_slice, response.url,
+ response.encoding)
+
diff --git a/scrapy/contrib_exp/crawlspider/reqgen.py b/scrapy/contrib_exp/crawlspider/reqgen.py
new file mode 100644
index 000000000..3858fbcf7
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/reqgen.py
@@ -0,0 +1,27 @@
+"""Request Generator"""
+from itertools import imap
+
+class RequestGenerator(object):
+ """Extracto and process requests from response"""
+
+ def __init__(self, req_extractors, req_processors, callback, spider=None):
+ """Initialize attributes"""
+ self._request_extractors = req_extractors
+ self._request_processors = req_processors
+ #TODO: resolve callback?
+ self._callback = callback
+
+ def generate_requests(self, response):
+ """Extract and process new requests from response.
+ Attach callback to each request as default callback."""
+ requests = []
+ for ext in self._request_extractors:
+ requests.extend(ext.extract_requests(response))
+
+ for proc in self._request_processors:
+ requests = proc(requests)
+
+ # return iterator
+ # @@@ creates new Request object with callback
+ return imap(lambda r: r.replace(callback=self._callback), requests)
+
diff --git a/scrapy/contrib_exp/crawlspider/reqproc.py b/scrapy/contrib_exp/crawlspider/reqproc.py
new file mode 100644
index 000000000..d39399b20
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/reqproc.py
@@ -0,0 +1,111 @@
+"""Request Processors"""
+from scrapy.utils.misc import arg_to_iter
+from scrapy.utils.url import canonicalize_url, url_is_from_any_domain
+
+from itertools import ifilter, imap
+
+import re
+
+class Canonicalize(object):
+ """Canonicalize Request Processor"""
+ def _replace_url(self, req):
+ # replace in-place
+ req.url = canonicalize_url(req.url)
+ return req
+
+ def __call__(self, requests):
+ """Canonicalize all requests' urls"""
+ return imap(self._replace_url, requests)
+
+
+class FilterDupes(object):
+ """Filter duplicate Requests"""
+
+ def __init__(self, *attributes):
+ """Initialize comparison attributes"""
+ self._attributes = tuple(attributes) if attributes \
+ else tuple(['url'])
+
+ def _equal_attr(self, obj1, obj2, attr):
+ return getattr(obj1, attr) == getattr(obj2, attr)
+
+ def _requests_equal(self, req1, req2):
+ """Attribute comparison helper"""
+ # look for not equal attribute
+ _not_equal = lambda attr: not self._equal_attr(req1, req2, attr)
+ for attr in ifilter(_not_equal, self._attributes):
+ return False
+ # all attributes equal
+ return True
+
+ def _request_in(self, request, requests_seen):
+ """Check if request is in given requests seen list"""
+ _req_seen = lambda r: self._requests_equal(r, request)
+ for seen in ifilter(_req_seen, requests_seen):
+ return True
+ # request not seen
+ return False
+
+ def __call__(self, requests):
+ """Filter seen requests"""
+ # per-call duplicates filter
+ self.requests_seen = set()
+ _not_seen = lambda r: not self._request_in(r, self.requests_seen)
+ for req in ifilter(_not_seen, requests):
+ yield req
+ # registry seen request
+ self.requests_seen.add(req)
+
+
+class FilterDomain(object):
+ """Filter request's domain"""
+
+ def __init__(self, allow=(), deny=()):
+ """Initialize allow/deny attributes"""
+ self.allow = tuple(arg_to_iter(allow))
+ self.deny = tuple(arg_to_iter(deny))
+
+ def __call__(self, requests):
+ """Filter domains"""
+ processed = (req for req in requests)
+
+ if self.allow:
+ processed = (req for req in requests
+ if url_is_from_any_domain(req.url, self.allow))
+ if self.deny:
+ processed = (req for req in requests
+ if not url_is_from_any_domain(req.url, self.deny))
+
+ return processed
+
+
+class FilterUrl(object):
+ """Filter request's url"""
+
+ def __init__(self, allow=(), deny=()):
+ """Initialize allow/deny attributes"""
+ _re_type = type(re.compile('', 0))
+
+ self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
+ for x in arg_to_iter(allow)]
+ self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
+ for x in arg_to_iter(deny)]
+
+ def __call__(self, requests):
+ """Filter request's url based on allow/deny rules"""
+ #TODO: filter valid urls here?
+ processed = (req for req in requests)
+
+ if self.allow_res:
+ processed = (req for req in requests
+ if self._matches(req.url, self.allow_res))
+ if self.deny_res:
+ processed = (req for req in requests
+ if not self._matches(req.url, self.deny_res))
+
+ return processed
+
+ def _matches(self, url, regexs):
+ """Returns True if url matches any regex in given list"""
+ return any(r.search(url) for r in regexs)
+
diff --git a/scrapy/contrib_exp/crawlspider/rules.py b/scrapy/contrib_exp/crawlspider/rules.py
new file mode 100644
index 000000000..ff1691ad0
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/rules.py
@@ -0,0 +1,100 @@
+"""Crawler Rules"""
+from scrapy.http import Request
+from scrapy.http import Response
+
+from functools import partial
+from itertools import ifilter
+
+from .matchers import BaseMatcher
+# default strint-to-matcher class
+from .matchers import UrlRegexMatcher
+
+class CompiledRule(object):
+ """Compiled version of Rule"""
+ def __init__(self, matcher, callback=None, follow=False):
+ """Initialize attributes checking type"""
+ assert isinstance(matcher, BaseMatcher)
+ assert callback is None or callable(callback)
+ assert isinstance(follow, bool)
+
+ self.matcher = matcher
+ self.callback = callback
+ self.follow = follow
+
+
+class Rule(object):
+ """Crawler Rule"""
+ def __init__(self, matcher=None, callback=None, follow=False, **kwargs):
+ """Store attributes"""
+ self.matcher = matcher
+ self.callback = callback
+ self.cb_kwargs = kwargs if kwargs else {}
+ self.follow = True if follow else False
+
+ if self.callback is None and self.follow is False:
+ raise ValueError("Rule must either have a callback or "
+ "follow=True: %r" % self)
+
+ def __repr__(self):
+ return "Rule(matcher=%r, callback=%r, follow=%r, **%r)" \
+ % (self.matcher, self.callback, self.follow, self.cb_kwargs)
+
+
+class RulesManager(object):
+ """Rules Manager"""
+ def __init__(self, rules, spider, default_matcher=UrlRegexMatcher):
+ """Initialize rules using spider and default matcher"""
+ self._rules = tuple()
+
+ # compile absolute/relative-to-spider callbacks"""
+ for rule in rules:
+ # prepare matcher
+ if rule.matcher is None:
+ # instance BaseMatcher by default
+ matcher = BaseMatcher()
+ elif isinstance(rule.matcher, BaseMatcher):
+ matcher = rule.matcher
+ else:
+ # matcher not BaseMatcher, check for string
+ if isinstance(rule.matcher, basestring):
+ # instance default matcher
+ matcher = default_matcher(rule.matcher)
+ else:
+ raise ValueError('Not valid matcher given %r in %r' \
+ % (rule.matcher, rule))
+
+ # prepare callback
+ if callable(rule.callback):
+ callback = rule.callback
+ elif not rule.callback is None:
+ # callback from spider
+ callback = getattr(spider, rule.callback)
+
+ if not callable(callback):
+ raise AttributeError('Invalid callback %r can not be resolved' \
+ % callback)
+ else:
+ callback = None
+
+ if rule.cb_kwargs:
+ # build partial callback
+ callback = partial(callback, **rule.cb_kwargs)
+
+ # append compiled rule to rules list
+ crule = CompiledRule(matcher, callback, follow=rule.follow)
+ self._rules += (crule, )
+
+ def get_rule_from_request(self, request):
+ """Returns first rule that matches given Request"""
+ _matches = lambda r: r.matcher.matches_request(request)
+ for rule in ifilter(_matches, self._rules):
+ # return first match of iterator
+ return rule
+
+ def get_rule_from_response(self, response):
+ """Returns first rule that matches given Response"""
+ _matches = lambda r: r.matcher.matches_response(response)
+ for rule in ifilter(_matches, self._rules):
+ # return first match of iterator
+ return rule
+
diff --git a/scrapy/contrib_exp/crawlspider/spider.py b/scrapy/contrib_exp/crawlspider/spider.py
new file mode 100644
index 000000000..730ad0e8d
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/spider.py
@@ -0,0 +1,69 @@
+"""CrawlSpider v2"""
+from scrapy.spider import BaseSpider
+from scrapy.utils.spider import iterate_spider_output
+
+from .matchers import UrlListMatcher
+from .rules import Rule, RulesManager
+from .reqext import SgmlRequestExtractor
+from .reqgen import RequestGenerator
+from .reqproc import Canonicalize, FilterDupes
+
+class CrawlSpider(BaseSpider):
+ """CrawlSpider v2"""
+
+ request_extractors = None
+ request_processors = None
+ rules = []
+
+ def __init__(self):
+ """Initialize dispatcher"""
+ super(CrawlSpider, self).__init__()
+
+ # auto follow start urls
+ if self.start_urls:
+ _matcher = UrlListMatcher(self.start_urls)
+ # append new rule using type from current self.rules
+ rules = self.rules + type(self.rules)([
+ Rule(_matcher, follow=True)
+ ])
+ else:
+ rules = self.rules
+
+ # set defaults if not set
+ if self.request_extractors is None:
+ # default link extractor. Extracts all links from response
+ self.request_extractors = [ SgmlRequestExtractor() ]
+
+ if self.request_processors is None:
+ # default proccessor. Filter duplicates requests
+ self.request_processors = [ FilterDupes() ]
+
+
+ # wrap rules
+ self._rulesman = RulesManager(rules, spider=self)
+ # generates new requests with given callback
+ self._reqgen = RequestGenerator(self.request_extractors,
+ self.request_processors,
+ callback=self.parse)
+
+ def parse(self, response):
+ """Dispatch callback and generate requests"""
+ # get rule for response
+ rule = self._rulesman.get_rule_from_response(response)
+
+ if rule:
+ # dispatch callback if set
+ if rule.callback:
+ output = iterate_spider_output(rule.callback(response))
+ for req_or_item in output:
+ yield req_or_item
+
+ if rule.follow:
+ for req in self._reqgen.generate_requests(response):
+ # only dispatch request if has matching rule
+ if self._rulesman.get_rule_from_request(req):
+ yield req
+ else:
+ self.log("No rule for response %s" % response, level=log.WARNING)
+
+
diff --git a/scrapy/contrib_exp/pipeline/shoveitem.py b/scrapy/contrib_exp/pipeline/shoveitem.py
deleted file mode 100644
index 869c62c58..000000000
--- a/scrapy/contrib_exp/pipeline/shoveitem.py
+++ /dev/null
@@ -1,55 +0,0 @@
-"""
-A pipeline to persist objects using shove.
-
-Shove is a "new generation" shelve. For more information see:
-http://pypi.python.org/pypi/shove
-"""
-
-from string import Template
-
-from shove import Shove
-from scrapy.xlib.pydispatch import dispatcher
-
-from scrapy import log
-from scrapy.core import signals
-from scrapy.conf import settings
-from scrapy.core.exceptions import NotConfigured
-
-class ShoveItemPipeline(object):
-
- def __init__(self):
- self.uritpl = settings['SHOVEITEM_STORE_URI']
- if not self.uritpl:
- raise NotConfigured
- self.opts = settings['SHOVEITEM_STORE_OPT'] or {}
- self.stores = {}
-
- dispatcher.connect(self.spider_opened, signal=signals.spider_opened)
- dispatcher.connect(self.spider_closed, signal=signals.spider_closed)
-
- def process_item(self, spider, item):
- guid = str(item.guid)
-
- if guid in self.stores[spider]:
- if self.stores[spider][guid] == item:
- status = 'old'
- else:
- status = 'upd'
- else:
- status = 'new'
-
- if not status == 'old':
- self.stores[spider][guid] = item
- self.log(spider, item, status)
- return item
-
- def spider_opened(self, spider):
- uri = Template(self.uritpl).substitute(domain=spider.domain_name)
- self.stores[spider] = Shove(uri, **self.opts)
-
- def spider_closed(self, spider):
- self.stores[spider].sync()
-
- def log(self, spider, item, status):
- log.msg("Shove (%s): Item guid=%s" % (status, item.guid), level=log.DEBUG, \
- spider=spider)
diff --git a/scrapy/core/downloader/manager.py b/scrapy/core/downloader/manager.py
index b53db5811..aec17b05e 100644
--- a/scrapy/core/downloader/manager.py
+++ b/scrapy/core/downloader/manager.py
@@ -2,6 +2,7 @@
Download web pages using asynchronous IO
"""
+import random
from time import time
from twisted.internet import reactor, defer
@@ -20,15 +21,21 @@ class SpiderInfo(object):
def __init__(self, download_delay=None, max_concurrent_requests=None):
if download_delay is None:
- self.download_delay = settings.getfloat('DOWNLOAD_DELAY')
+ self._download_delay = settings.getfloat('DOWNLOAD_DELAY')
else:
- self.download_delay = download_delay
- if self.download_delay:
+ self._download_delay = float(download_delay)
+ if self._download_delay:
self.max_concurrent_requests = 1
elif max_concurrent_requests is None:
self.max_concurrent_requests = settings.getint('CONCURRENT_REQUESTS_PER_SPIDER')
else:
self.max_concurrent_requests = max_concurrent_requests
+ if self._download_delay and settings.getbool('RANDOMIZE_DOWNLOAD_DELAY'):
+ # same policy as wget --random-wait
+ self.random_delay_interval = (0.5*self._download_delay, \
+ 1.5*self._download_delay)
+ else:
+ self.random_delay_interval = None
self.active = set()
self.queue = []
@@ -44,6 +51,12 @@ class SpiderInfo(object):
# use self.active to include requests in the downloader middleware
return len(self.active) > 2 * self.max_concurrent_requests
+ def download_delay(self):
+ if self.random_delay_interval:
+ return random.uniform(*self.random_delay_interval)
+ else:
+ return self._download_delay
+
def cancel_request_calls(self):
for call in self.next_request_calls:
call.cancel()
@@ -99,8 +112,9 @@ class Downloader(object):
# Delay queue processing if a download_delay is configured
now = time()
- if site.download_delay:
- penalty = site.download_delay - now + site.lastseen
+ delay = site.download_delay()
+ if delay:
+ penalty = delay - now + site.lastseen
if penalty > 0:
d = defer.Deferred()
d.addCallback(self._process_queue)
diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py
index 1c13e729c..d42a89197 100644
--- a/scrapy/http/response/text.py
+++ b/scrapy/http/response/text.py
@@ -5,6 +5,7 @@ discovering (through HTTP headers) to base Response class.
See documentation in docs/topics/request-response.rst
"""
+import codecs
import re
from scrapy.xlib.BeautifulSoup import UnicodeDammit
@@ -64,7 +65,12 @@ class TextResponse(Response):
if content_type:
encoding = self._ENCODING_RE.search(content_type)
if encoding:
- return encoding.group(1)
+ enc = encoding.group(1)
+ try:
+ codecs.lookup(enc) # check if the encoding is valid
+ return enc
+ except LookupError:
+ pass
@memoizemethod_noargs
def body_as_unicode(self):
diff --git a/scrapy/templates/project/module/pipelines.py.tmpl b/scrapy/templates/project/module/pipelines.py.tmpl
index fa6f5ea6f..e3f89342d 100644
--- a/scrapy/templates/project/module/pipelines.py.tmpl
+++ b/scrapy/templates/project/module/pipelines.py.tmpl
@@ -4,5 +4,5 @@
# See: http://doc.scrapy.org/topics/item-pipeline.html
class ${ProjectName}Pipeline(object):
- def process_item(self, domain, item):
+ def process_item(self, spider, item):
return item
diff --git a/scrapy/templates/spiders/crawl.tmpl b/scrapy/templates/spiders/crawl.tmpl
index 2d4f7fce6..18fd0ec42 100644
--- a/scrapy/templates/spiders/crawl.tmpl
+++ b/scrapy/templates/spiders/crawl.tmpl
@@ -10,15 +10,15 @@ class $classname(CrawlSpider):
start_urls = ['http://www.$site/']
rules = (
- Rule(SgmlLinkExtractor(allow=(r'Items/', )), 'parse_item', follow=True),
+ Rule(SgmlLinkExtractor(allow=r'Items/'), callback='parse_item', follow=True),
)
def parse_item(self, response):
- xs = HtmlXPathSelector(response)
+ hxs = HtmlXPathSelector(response)
i = ${ProjectName}Item()
- #i['site_id'] = xs.select('//input[@id="sid"]/@value').extract()
- #i['name'] = xs.select('//div[@id="name"]').extract()
- #i['description'] = xs.select('//div[@id="description"]').extract()
+ #i['site_id'] = hxs.select('//input[@id="sid"]/@value').extract()
+ #i['name'] = hxs.select('//div[@id="name"]').extract()
+ #i['description'] = hxs.select('//div[@id="description"]').extract()
return i
SPIDER = $classname()
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_matchers.py b/scrapy/tests/test_contrib_exp_crawlspider_matchers.py
new file mode 100644
index 000000000..4cb832aa1
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_matchers.py
@@ -0,0 +1,94 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import Response
+
+from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher
+
+import re
+
+class MatchersTest(unittest.TestCase):
+
+ def setUp(self):
+ pass
+
+ def test_base_matcher(self):
+ matcher = BaseMatcher()
+
+ request = Request('http://example.com')
+ response = Response('http://example.com')
+
+ self.assertTrue(matcher.matches_request(request))
+ self.assertTrue(matcher.matches_response(response))
+
+ def test_url_matcher(self):
+ matcher = UrlMatcher('http://example.com')
+
+ request = Request('http://example.com')
+ response = Response('http://example.com')
+
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_request(response))
+
+ request = Request('http://example2.com')
+ response = Response('http://example2.com')
+
+ self.failIf(matcher.matches_request(request))
+ self.failIf(matcher.matches_request(response))
+
+ def test_url_regex_matcher(self):
+ matcher = UrlRegexMatcher(r'sample')
+ urls = (
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample4.html',
+ )
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_response(response))
+
+ matcher = UrlRegexMatcher(r'sample_fail')
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failIf(matcher.matches_request(request))
+ self.failIf(matcher.matches_response(response))
+
+ matcher = UrlRegexMatcher(r'SAMPLE\d+', re.IGNORECASE)
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_response(response))
+
+ def test_url_list_matcher(self):
+ urls = (
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample4.html',
+ )
+ urls2 = (
+ 'http://example.com/sample5.html',
+ 'http://example.com/sample6.html',
+ 'http://example.com/sample7.html',
+ 'http://example.com/sample8.html',
+ 'http://example.com/',
+ )
+ matcher = UrlListMatcher(urls)
+
+ # match urls
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_response(response))
+
+ # non-match urls
+ for url in urls2:
+ request, response = Request(url), Response(url)
+ self.failIf(matcher.matches_request(request))
+ self.failIf(matcher.matches_response(response))
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqext.py b/scrapy/tests/test_contrib_exp_crawlspider_reqext.py
new file mode 100644
index 000000000..d0d0d1b5f
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_reqext.py
@@ -0,0 +1,137 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import HtmlResponse
+from scrapy.tests import get_testdata
+
+from scrapy.contrib_exp.crawlspider.reqext import BaseSgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqext import XPathRequestExtractor
+
+class AbstractRequestExtractorTest(unittest.TestCase):
+
+ def _requests_equals(self, list1, list2):
+ """Compares request's urls and link_text"""
+ for (r1, r2) in zip(list1, list2):
+ if r1.url != r2.url:
+ return False
+ if r1.meta['link_text'] != r2.meta['link_text']:
+ return False
+ # all equal
+ return True
+
+
+class RequestExtractorTest(AbstractRequestExtractorTest):
+
+ def test_basic(self):
+ base_url = 'http://example.org/somepage/index.html'
+ html = """
Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+ requests = [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ Request('http://example.org/about.html',
+ meta={'link_text': 'About us'}),
+ Request('http://example.org/othercat.html',
+ meta={'link_text': 'Other category'}),
+ Request('http://example.org/',
+ meta={'link_text': ''}),
+ ]
+
+ response = HtmlResponse(base_url, body=html)
+ reqx = BaseSgmlRequestExtractor() # default: tag=a, attr=href
+
+ self.failUnless(
+ self._requests_equals(requests, reqx.extract_requests(response))
+ )
+
+ def test_base_url(self):
+ html = """Page title
+
+ Item 12
+ """
+ response = HtmlResponse("http://example.org/somepage/index.html",
+ body=html)
+ reqx = BaseSgmlRequestExtractor()
+
+ self.failUnless(
+ self._requests_equals(reqx.extract_requests(response),
+ [ Request('http://otherdomain.com/base/item/12.html',
+ meta={'link_text': 'Item 12'}) ]
+ )
+ )
+
+ def test_extraction_encoding(self):
+ #TODO: use own fixtures
+ body = get_testdata('link_extractor', 'linkextractor_noenc.html')
+ response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body,
+ headers={'Content-Type': ['text/html; charset=utf-8']})
+ response_noenc = HtmlResponse(url='http://example.com/noenc',
+ body=body)
+ body = get_testdata('link_extractor', 'linkextractor_latin1.html')
+ response_latin1 = HtmlResponse(url='http://example.com/latin1',
+ body=body)
+
+ reqx = BaseSgmlRequestExtractor()
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(response_utf8),
+ [ Request(url='http://example.com/sample_%C3%B1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample_%E2%82%AC.html',
+ meta={'link_text':
+ 'sample \xe2\x82\xac text'.decode('utf-8')}) ]
+ )
+ )
+
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(response_noenc),
+ [ Request(url='http://example.com/sample_%C3%B1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample_%E2%82%AC.html',
+ meta={'link_text':
+ 'sample \xe2\x82\xac text'.decode('utf-8')}) ]
+ )
+ )
+
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(response_latin1),
+ [ Request(url='http://example.com/sample_%F1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample_%E1.html',
+ meta={'link_text':
+ 'sample \xe1 text'.decode('latin1')}) ]
+ )
+ )
+
+
+class SgmlRequestExtractorTest(AbstractRequestExtractorTest):
+ pass
+
+
+class XPathRequestExtractorTest(AbstractRequestExtractorTest):
+
+ def setUp(self):
+ # TODO: use own fixtures
+ body = get_testdata('link_extractor', 'sgml_linkextractor.html')
+ self.response = HtmlResponse(url='http://example.com/index', body=body)
+
+
+ def test_restrict_xpaths(self):
+ reqx = XPathRequestExtractor('//div[@id="subwrapper"]')
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(self.response),
+ [ Request(url='http://example.com/sample1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample2.html',
+ meta={'link_text': 'sample 2'}) ]
+ )
+ )
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py b/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py
new file mode 100644
index 000000000..67aca2387
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py
@@ -0,0 +1,128 @@
+from twisted.internet import defer
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import HtmlResponse
+from scrapy.utils.python import equal_attributes
+
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqgen import RequestGenerator
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain
+from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
+
+class RequestGeneratorTest(unittest.TestCase):
+
+ def setUp(self):
+ url = 'http://example.org/somepage/index.html'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+
+ self.response = HtmlResponse(url, body=html)
+ self.deferred = defer.Deferred()
+ self.requests = [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ Request('http://example.org/about.html',
+ meta={'link_text': 'About us'}),
+ Request('http://example.org/othercat.html',
+ meta={'link_text': 'Other category'}),
+ Request('http://example.org/',
+ meta={'link_text': ''}),
+ ]
+
+ def _equal_requests_list(self, list1, list2):
+ list1 = list(list1)
+ list2 = list(list2)
+ if not len(list1) == len(list2):
+ return False
+
+ for (req1, req2) in zip(list1, list2):
+ if not equal_attributes(req1, req2, ['url']):
+ return False
+ return True
+
+ def test_basic(self):
+ reqgen = RequestGenerator([], [], callback=self.deferred)
+ # returns generator
+ requests = reqgen.generate_requests(self.response)
+ self.failUnlessEqual(list(requests), [])
+
+ def test_request_extractor(self):
+ extractors = [
+ SgmlRequestExtractor()
+ ]
+
+ # extract all requests
+ reqgen = RequestGenerator(extractors, [], callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+ self.failUnless(self._equal_requests_list(requests, self.requests))
+
+ for req in requests:
+ # check callback
+ self.failUnlessEqual(req.deferred, self.deferred)
+
+ def test_request_processor(self):
+ extractors = [
+ SgmlRequestExtractor()
+ ]
+
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+ self.failUnless(self._equal_requests_list(requests, self.requests))
+
+ # filter domain
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterDomain(deny='example.org'),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+ self.failUnlessEqual(list(requests), [])
+
+ # filter url
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterUrl(deny=(r'about', r'othercat')),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+
+ self.failUnless(self._equal_requests_list(requests, [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ Request('http://example.org/',
+ meta={'link_text': ''}),
+ ]))
+
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterUrl(allow=r'/somepage/'),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+
+ self.failUnless(self._equal_requests_list(requests, [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ ]))
+
+
+
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py b/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py
new file mode 100644
index 000000000..da5db67b2
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py
@@ -0,0 +1,144 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain
+from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
+
+import copy
+
+class RequestProcessorsTest(unittest.TestCase):
+
+ def test_canonicalize_requests(self):
+ urls = [
+ 'http://example.com/do?&b=1&a=2&c=3',
+ 'http://example.com/do?123,&q=a space',
+ ]
+ urls_after = [
+ 'http://example.com/do?a=2&b=1&c=3',
+ 'http://example.com/do?123%2C=&q=a+space',
+ ]
+
+ proc = Canonicalize()
+ results = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(results, urls_after)
+
+ def test_unique_requests(self):
+ urls = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ ]
+ urls_unique = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ ]
+
+ proc = FilterDupes()
+ results = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(results, urls_unique)
+
+ # Check custom attributes
+ requests = [
+ Request('http://example.com', method='GET'),
+ Request('http://example.com', method='POST'),
+ ]
+ proc = FilterDupes('url', 'method')
+ self.failUnlessEqual(len(list(proc(requests))), 2)
+
+ proc = FilterDupes('url')
+ self.failUnlessEqual(len(list(proc(requests))), 1)
+
+ def test_filter_domain(self):
+ urls = [
+ 'http://blah1.com/index',
+ 'http://blah2.com/index',
+ 'http://blah1.com/section',
+ 'http://blah2.com/section',
+ ]
+
+ proc = FilterDomain(allow=('blah1.com'), deny=('blah2.com'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [
+ 'http://blah1.com/index',
+ 'http://blah1.com/section',
+ ])
+
+ proc = FilterDomain(deny=('blah1.com', 'blah2.com'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [])
+
+ proc = FilterDomain(allow=('blah1.com', 'blah2.com'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, urls)
+
+ def test_filter_url(self):
+ urls = [
+ 'http://blah1.com/index',
+ 'http://blah2.com/index',
+ 'http://blah1.com/section',
+ 'http://blah2.com/section',
+ ]
+
+ proc = FilterUrl(allow=(r'blah1'), deny=(r'blah2'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [
+ 'http://blah1.com/index',
+ 'http://blah1.com/section',
+ ])
+
+ proc = FilterUrl(deny=('blah1', 'blah2'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [])
+
+ proc = FilterUrl(allow=('index$', 'section$'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, urls)
+
+
+
+ def test_all_processors(self):
+ urls = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/do?&b=1&a=2&c=3',
+ 'http://example.com/do?123,&q=a space',
+ ]
+ urls_processed = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/do?a=2&b=1&c=3',
+ 'http://example.com/do?123%2C=&q=a+space',
+ ]
+
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ ]
+
+ def _process(requests):
+ """Apply all processors"""
+ # copy list
+ processed = [copy.copy(req) for req in requests]
+ for proc in processors:
+ processed = proc(processed)
+ return processed
+
+ # empty requests
+ results1 = [r.url for r in _process([])]
+ self.failUnlessEquals(results1, [])
+
+ # try urls
+ requests = (Request(url) for url in urls)
+ results2 = [r.url for r in _process(requests)]
+ self.failUnlessEquals(results2, urls_processed)
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_rules.py b/scrapy/tests/test_contrib_exp_crawlspider_rules.py
new file mode 100644
index 000000000..0fbe52415
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_rules.py
@@ -0,0 +1,262 @@
+from twisted.trial import unittest
+
+from scrapy.http import HtmlResponse
+from scrapy.spider import BaseSpider
+from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
+
+from scrapy.contrib_exp.crawlspider.rules import CompiledRule
+from scrapy.contrib_exp.crawlspider.rules import Rule
+from scrapy.contrib_exp.crawlspider.rules import RulesManager
+
+from functools import partial
+
+class RuleInitializationTest(unittest.TestCase):
+
+ def test_fail_if_rule_null(self):
+ # fail on empty rule
+ self.failUnlessRaises(ValueError, Rule)
+ self.failUnlessRaises(ValueError, Rule,
+ **dict(callback=None, follow=None))
+ self.failUnlessRaises(ValueError, Rule,
+ **dict(callback=None, follow=False))
+
+ def test_minimal_arguments_to_instantiation(self):
+ # not fail if callback set
+ self.failUnless(Rule(callback=lambda: True))
+ # not fail if follow set
+ self.failUnless(Rule(follow=True))
+
+ def test_validate_default_attributes(self):
+ # test null Rule
+ rule = Rule(follow=True)
+ self.failUnlessEqual(None, rule.matcher)
+ self.failUnlessEqual(None, rule.callback)
+ self.failUnlessEqual({}, rule.cb_kwargs)
+ # follow default False
+ self.failUnlessEqual(True, rule.follow)
+
+ def test_validate_attributes_set(self):
+ matcher = BaseMatcher()
+ callback = lambda: True
+ rule = Rule(matcher, callback, True, a=1)
+ # test attributes
+ self.failUnlessEqual(matcher, rule.matcher)
+ self.failUnlessEqual(callback, rule.callback)
+ self.failUnlessEqual({'a': 1}, rule.cb_kwargs)
+ self.failUnlessEqual(True, rule.follow)
+
+class CompiledRuleInitializationTest(unittest.TestCase):
+
+ def test_fail_on_invalid_matcher(self):
+ # pass with valid matcher
+ self.failUnless(CompiledRule(BaseMatcher()),
+ "Failed CompiledRule instantiation")
+
+ # at least needs valid matcher
+ self.assertRaises(AssertionError, CompiledRule, None)
+ self.assertRaises(AssertionError, CompiledRule, False)
+ self.assertRaises(AssertionError, CompiledRule, True)
+
+ def test_fail_on_invalid_callback(self):
+ # pass with valid callback
+ callback = lambda: True
+ self.failUnless(CompiledRule(BaseMatcher(), callback))
+ # pass with callback none
+ self.failUnless(CompiledRule(BaseMatcher(), None))
+
+ # assert on invalid callback
+ self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
+ 'myfunc')
+
+ # numeric variable
+ var = 123
+ self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
+ var)
+
+ class A:
+ pass
+
+ # random instance
+ self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
+ A())
+
+
+ def test_fail_on_invalid_follow_value(self):
+ callback = lambda: True
+ matcher = BaseMatcher()
+ # pass bool
+ self.failUnless(CompiledRule(matcher, callback, True))
+ self.failUnless(CompiledRule(matcher, callback, False))
+
+ # assert with non-bool
+ self.assertRaises(AssertionError, CompiledRule, matcher,
+ callback, None)
+ self.assertRaises(AssertionError, CompiledRule, matcher,
+ callback, 1)
+
+ def test_validate_default_attributes(self):
+ callback = lambda: True
+ matcher = BaseMatcher()
+ rule = CompiledRule(matcher, callback, True)
+
+ # test attributes
+ self.failUnlessEqual(matcher, rule.matcher)
+ self.failUnlessEqual(callback, rule.callback)
+ self.failUnlessEqual(True, rule.follow)
+
+
+class RulesTest(unittest.TestCase):
+ def test_rules_manager_basic(self):
+ spider = BaseSpider()
+ response1 = HtmlResponse('http://example.org')
+ response2 = HtmlResponse('http://othersite.org')
+ rulesman = RulesManager([], spider)
+
+ # should return none
+ self.failIf(rulesman.get_rule_from_response(response1))
+ self.failIf(rulesman.get_rule_from_response(response2))
+
+ # rules manager with match-all rule
+ rulesman = RulesManager([
+ Rule(BaseMatcher(), follow=True),
+ ], spider)
+
+ # returns CompiledRule
+ rule1 = rulesman.get_rule_from_response(response1)
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnless(isinstance(rule1, CompiledRule))
+ self.failUnless(isinstance(rule2, CompiledRule))
+ self.assert_(rule1 is rule2)
+ self.failUnlessEqual(rule1.callback, None)
+ self.failUnlessEqual(rule1.follow, True)
+
+ def test_rules_manager_empty_rule(self):
+ spider = BaseSpider()
+ response = HtmlResponse('http://example.org')
+
+ rulesman = RulesManager([Rule(follow=True)], spider)
+
+ rule = rulesman.get_rule_from_response(response)
+ # default matcher if None: BaseMatcher
+ self.failUnless(isinstance(rule.matcher, BaseMatcher))
+
+ def test_rules_manager_default_matcher(self):
+ spider = BaseSpider()
+ response = HtmlResponse('http://example.org')
+ callback = lambda x: None
+
+ rulesman = RulesManager([
+ Rule('http://example.org', callback),
+ ], spider, default_matcher=UrlMatcher)
+
+ rule = rulesman.get_rule_from_response(response)
+ self.failUnless(isinstance(rule.matcher, UrlMatcher))
+
+ def test_rules_manager_matchers(self):
+ spider = BaseSpider()
+ response1 = HtmlResponse('http://example.org')
+ response2 = HtmlResponse('http://othersite.org')
+
+ urlmatcher = UrlMatcher('http://example.org')
+ basematcher = BaseMatcher()
+ # callback needed for Rule
+ callback = lambda x: None
+
+ # test fail matcher resolve
+ self.assertRaises(ValueError, RulesManager,
+ [Rule(False, callback)], spider)
+ self.assertRaises(ValueError, RulesManager,
+ [Rule(spider, callback)], spider)
+
+ rulesman = RulesManager([
+ Rule(urlmatcher, callback),
+ Rule(basematcher, callback),
+ ], spider)
+
+ # response1 matches example.org
+ rule1 = rulesman.get_rule_from_response(response1)
+ # response2 is catch by BaseMatcher()
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnlessEqual(rule1.matcher, urlmatcher)
+ self.failUnlessEqual(rule2.matcher, basematcher)
+
+ # reverse order. BaseMatcher should match all
+ rulesman = RulesManager([
+ Rule(basematcher, callback),
+ Rule(urlmatcher, callback),
+ ], spider)
+
+ rule1 = rulesman.get_rule_from_response(response1)
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnlessEqual(rule1.matcher, basematcher)
+ self.failUnlessEqual(rule2.matcher, basematcher)
+ self.failUnless(rule1 is rule2)
+
+ def test_rules_manager_callbacks(self):
+ mycallback = lambda: True
+
+ spider = BaseSpider()
+ spider.parse_item = lambda: True
+
+ response1 = HtmlResponse('http://example.org')
+ response2 = HtmlResponse('http://othersite.org')
+
+ rulesman = RulesManager([
+ Rule('example', mycallback),
+ Rule('othersite', 'parse_item'),
+ ], spider, default_matcher=UrlRegexMatcher)
+
+ rule1 = rulesman.get_rule_from_response(response1)
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnlessEqual(rule1.callback, mycallback)
+ self.failUnlessEqual(rule2.callback, spider.parse_item)
+
+ # fail unknown callback
+ self.assertRaises(AttributeError, RulesManager, [
+ Rule(BaseMatcher(), 'mycallback')
+ ], spider)
+ # fail not callable
+ spider.not_callable = True
+ self.assertRaises(AttributeError, RulesManager, [
+ Rule(BaseMatcher(), 'not_callable')
+ ], spider)
+
+
+ def test_rules_manager_callback_with_arguments(self):
+ spider = BaseSpider()
+ response = HtmlResponse('http://example.org')
+
+ kwargs = {'a': 1}
+
+ def myfunc(**mykwargs):
+ return mykwargs
+
+ # verify return validation
+ self.failUnlessEquals(kwargs, myfunc(**kwargs))
+
+ # test callback w/o arguments
+ rulesman = RulesManager([
+ Rule(BaseMatcher(), myfunc),
+ ], spider)
+ rule = rulesman.get_rule_from_response(response)
+
+ # without arguments should return same callback
+ self.failUnlessEqual(rule.callback, myfunc)
+
+ # test callback w/ arguments
+ rulesman = RulesManager([
+ Rule(BaseMatcher(), myfunc, **kwargs),
+ ], spider)
+ rule = rulesman.get_rule_from_response(response)
+
+ # with argument should return partial applied callback
+ self.failUnless(isinstance(rule.callback, partial))
+ self.failUnlessEquals(kwargs, rule.callback())
+
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_spider.py b/scrapy/tests/test_contrib_exp_crawlspider_spider.py
new file mode 100644
index 000000000..5e067508a
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_spider.py
@@ -0,0 +1,222 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import HtmlResponse
+from scrapy.item import BaseItem
+from scrapy.utils.spider import iterate_spider_output
+
+# basics
+from scrapy.contrib_exp.crawlspider import CrawlSpider
+from scrapy.contrib_exp.crawlspider import Rule
+
+# matchers
+from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher
+
+# extractors
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+
+# processors
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
+
+
+# mock items
+class Item1(BaseItem):
+ pass
+
+class Item2(BaseItem):
+ pass
+
+class Item3(BaseItem):
+ pass
+
+
+class CrawlSpiderTest(unittest.TestCase):
+
+ def spider_factory(self, rules=[],
+ extractors=[], processors=[],
+ start_urls=[]):
+ # mock spider
+ class Spider(CrawlSpider):
+ def parse_item1(self, response):
+ return Item1()
+
+ def parse_item2(self, response):
+ return Item2()
+
+ def parse_item3(self, response):
+ return Item3()
+
+ def parse_request1(self, response):
+ return Request('http://example.org/request1')
+
+ def parse_request2(self, response):
+ return Request('http://example.org/request2')
+
+ Spider.start_urls = start_urls
+ Spider.rules = rules
+ Spider.request_extractors = extractors
+ Spider.request_processors = processors
+
+ return Spider()
+
+ def test_start_url_auto_rule(self):
+ spider = self.spider_factory()
+ # zero spider rules
+ self.failUnlessEqual(len(spider.rules), 0)
+ self.failUnlessEqual(len(spider._rulesman._rules), 0)
+
+ spider = self.spider_factory(start_urls=['http://example.org'])
+
+ self.failUnlessEqual(len(spider.rules), 0)
+ self.failUnlessEqual(len(spider._rulesman._rules), 1)
+
+ def test_start_url_matcher(self):
+ url = 'http://example.org'
+ spider = self.spider_factory(start_urls=[url])
+
+ response = HtmlResponse(url)
+
+ rule = spider._rulesman.get_rule_from_response(response)
+ self.failUnless(isinstance(rule.matcher, UrlListMatcher))
+
+ response = HtmlResponse(url + '/item.html')
+
+ rule = spider._rulesman.get_rule_from_response(response)
+ self.failUnless(rule is None)
+
+ # TODO: remove this block
+ # in previous version get_rule returns rule from response.request
+ response.request = Request(url)
+ rule = spider._rulesman.get_rule_from_response(response.request)
+ self.failUnless(isinstance(rule.matcher, UrlListMatcher))
+ self.failUnlessEqual(rule.follow, True)
+
+ def test_parse_callback(self):
+ response = HtmlResponse('http://example.org')
+ rules = (
+ Rule(BaseMatcher(), 'parse_item1'),
+ )
+ spider = self.spider_factory(rules)
+
+ result = list(spider.parse(response))
+ self.failUnlessEqual(len(result), 1)
+ self.failUnless(isinstance(result[0], Item1))
+
+ def test_crawling_start_url(self):
+ url = 'http://example.org/'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+ response = HtmlResponse(url, body=html)
+
+ extractors = (SgmlRequestExtractor(), )
+ spider = self.spider_factory(start_urls=[url],
+ extractors=extractors)
+ result = list(spider.parse(response))
+
+ # 1 request extracted: example.org/
+ # because requests returns only matching
+ self.failUnlessEqual(len(result), 1)
+
+ # we will add catch-all rule to extract all
+ callback = lambda x: None
+ rules = [Rule(r'\.html$', callback=callback)]
+ spider = self.spider_factory(rules, start_urls=[url],
+ extractors=extractors)
+ result = list(spider.parse(response))
+
+ # 4 requests extracted
+ # 3 of .html pattern
+ # 1 of start url patter
+ self.failUnlessEqual(len(result), 4)
+
+ def test_crawling_simple_rule(self):
+ url = 'http://example.org/somepage/index.html'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+
+ response = HtmlResponse(url, body=html)
+
+ rules = (
+ # first response callback
+ Rule(r'index\.html', 'parse_item1'),
+ )
+ spider = self.spider_factory(rules)
+ result = list(spider.parse(response))
+
+ # should return Item1
+ self.failUnlessEqual(len(result), 1)
+ self.failUnless(isinstance(result[0], Item1))
+
+ # test request generation
+ rules = (
+ # first response without callback and follow flag
+ Rule(r'index\.html', follow=True),
+ Rule(r'(\.html|/)$', 'parse_item1'),
+ )
+ spider = self.spider_factory(rules)
+ result = list(spider.parse(response))
+
+ # 0 because spider does not have extractors
+ self.failUnlessEqual(len(result), 0)
+
+ extractors = (SgmlRequestExtractor(), )
+
+ # instance spider with extractor
+ spider = self.spider_factory(rules, extractors)
+ result = list(spider.parse(response))
+ # 4 requests extracted
+ self.failUnlessEqual(len(result), 4)
+
+ def test_crawling_multiple_rules(self):
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+
+ response = HtmlResponse('http://example.org/index.html', body=html)
+ response1 = HtmlResponse('http://example.org/1.html')
+ response2 = HtmlResponse('http://example.org/othercat.html')
+
+ rules = (
+ Rule(r'\d+\.html$', 'parse_item1'),
+ Rule(r'othercat\.html$', 'parse_item2'),
+ # follow-only rules
+ Rule(r'index\.html', 'parse_item3', follow=True)
+ )
+ extractors = [SgmlRequestExtractor()]
+ spider = self.spider_factory(rules, extractors)
+
+ result = list(spider.parse(response))
+ # 1 Item 2 Requests
+ self.failUnlessEqual(len(result), 3)
+ # parse_item3
+ self.failUnless(isinstance(result[0], Item3))
+ only_requests = lambda r: isinstance(r, Request)
+ requests = filter(only_requests, result[1:])
+ self.failUnlessEqual(len(requests), 2)
+ self.failUnless(all(requests))
+
+ result1 = list(spider.parse(response1))
+ # parse_item1
+ self.failUnlessEqual(len(result1), 1)
+ self.failUnless(isinstance(result1[0], Item1))
+
+ result2 = list(spider.parse(response2))
+ # parse_item2
+ self.failUnlessEqual(len(result2), 1)
+ self.failUnless(isinstance(result2[0], Item2))
+
+
diff --git a/scrapy/tests/test_http_response.py b/scrapy/tests/test_http_response.py
index 3b0a144d2..5851572ac 100644
--- a/scrapy/tests/test_http_response.py
+++ b/scrapy/tests/test_http_response.py
@@ -175,6 +175,8 @@ class TextResponseTest(BaseResponseTest):
r2 = self.response_class("http://www.example.com", encoding='utf-8', body=u"\xa3")
r3 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=iso-8859-1"]}, body="\xa3")
r4 = self.response_class("http://www.example.com", body="\xa2\xa3")
+ r5 = self.response_class("http://www.example.com",
+ headers={"Content-type": ["text/html; charset=None"]}, body="\xc2\xa3")
self.assertEqual(r1.headers_encoding(), "utf-8")
self.assertEqual(r2.headers_encoding(), None)
@@ -182,6 +184,8 @@ class TextResponseTest(BaseResponseTest):
self.assertEqual(r3.headers_encoding(), "iso-8859-1")
self.assertEqual(r3.encoding, 'iso-8859-1')
self.assertEqual(r4.headers_encoding(), None)
+ self.assertEqual(r5.headers_encoding(), None)
+ self.assertEqual(r5.encoding, "utf-8")
assert r4.body_encoding() is not None and r4.body_encoding() != 'ascii'
self._assert_response_values(r1, 'utf-8', u"\xa3")
self._assert_response_values(r2, 'utf-8', u"\xa3")
diff --git a/scrapy/tests/test_utils_python.py b/scrapy/tests/test_utils_python.py
index 13d8be30c..dc46e1f91 100644
--- a/scrapy/tests/test_utils_python.py
+++ b/scrapy/tests/test_utils_python.py
@@ -1,7 +1,8 @@
+import operator
import unittest
from scrapy.utils.python import str_to_unicode, unicode_to_str, \
- memoizemethod_noargs, isbinarytext
+ memoizemethod_noargs, isbinarytext, equal_attributes
class UtilsPythonTestCase(unittest.TestCase):
def test_str_to_unicode(self):
@@ -61,5 +62,52 @@ class UtilsPythonTestCase(unittest.TestCase):
# finally some real binary bytes
assert isbinarytext("\x02\xa3")
+ def test_equal_attributes(self):
+ class Obj:
+ pass
+
+ a = Obj()
+ b = Obj()
+ # no attributes given return False
+ self.failIf(equal_attributes(a, b, []))
+ # not existent attributes
+ self.failIf(equal_attributes(a, b, ['x', 'y']))
+
+ a.x = 1
+ b.x = 1
+ # equal attribute
+ self.failUnless(equal_attributes(a, b, ['x']))
+
+ b.y = 2
+ # obj1 has no attribute y
+ self.failIf(equal_attributes(a, b, ['x', 'y']))
+
+ a.y = 2
+ # equal attributes
+ self.failUnless(equal_attributes(a, b, ['x', 'y']))
+
+ a.y = 1
+ # differente attributes
+ self.failIf(equal_attributes(a, b, ['x', 'y']))
+
+ # test callable
+ a.meta = {}
+ b.meta = {}
+ self.failUnless(equal_attributes(a, b, ['meta']))
+
+ # compare ['meta']['a']
+ a.meta['z'] = 1
+ b.meta['z'] = 1
+
+ get_z = operator.itemgetter('z')
+ get_meta = operator.attrgetter('meta')
+ compare_z = lambda obj: get_z(get_meta(obj))
+
+ self.failUnless(equal_attributes(a, b, [compare_z, 'x']))
+ # fail z equality
+ a.meta['z'] = 2
+ self.failIf(equal_attributes(a, b, [compare_z, 'x']))
+
+
if __name__ == "__main__":
unittest.main()
diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py
index 7d43c4566..319acdc05 100644
--- a/scrapy/utils/python.py
+++ b/scrapy/utils/python.py
@@ -216,3 +216,28 @@ def get_func_args(func):
else:
raise TypeError('%s is not callable' % type(func))
return func_args
+
+
+def equal_attributes(obj1, obj2, attributes):
+ """Compare two objects attributes"""
+ # not attributes given return False by default
+ if not attributes:
+ return False
+
+ for attr in attributes:
+ # support callables like itemgetter
+ if callable(attr):
+ if not attr(obj1) == attr(obj2):
+ return False
+ else:
+ # check that objects has attribute
+ if not hasattr(obj1, attr):
+ return False
+ if not hasattr(obj2, attr):
+ return False
+ # compare object attributes
+ if not getattr(obj1, attr) == getattr(obj2, attr):
+ return False
+ # all attributes equal
+ return True
+