diff --git a/AUTHORS b/AUTHORS
index a0fbe722f..1392aa71f 100644
--- a/AUTHORS
+++ b/AUTHORS
@@ -1,28 +1,25 @@
Scrapy was brought to life by Shane Evans while hacking a scraping framework
prototype for Mydeco (mydeco.com). It soon became maintained, extended and
-improved by Insophia (insophia.com), with the sponsorship of By Design (the
-company behind Mydeco).
+improved by Insophia (insophia.com), with the initial sponsorship of Mydeco to
+bootstrap the project.
-Here is the list of the primary authors & contributors, along with their user
-name (in Scrapy trac/subversion). Emails are intentionally left out to avoid
-spam.
+Here is the list of the primary authors & contributors:
- * Pablo Hoffman (pablo)
- * Daniel Graña (daniel)
- * Martin Olveyra (olveyra)
- * Gabriel García (elpolilla)
- * Michael Cetrulo (samus_)
- * Artem Bogomyagkov (artem)
- * Damian Canabal (calarval)
- * Andres Moreira (andres)
- * Ismael Carnales (ismael)
- * Matías Aguirre (omab)
- * German Hoffman (german)
- * Anibal Pacheco (anibal)
+ * Pablo Hoffman
+ * Daniel Graña
+ * Martin Olveyra
+ * Gabriel García
+ * Michael Cetrulo
+ * Artem Bogomyagkov
+ * Damian Canabal
+ * Andres Moreira
+ * Ismael Carnales
+ * Matías Aguirre
+ * German Hoffmann
+ * Anibal Pacheco
* Bruno Deferrari
* Shane Evans
-
-And here is the list of people who have helped to put the Scrapy homepage live:
-
- * Ezequiel Rivero (ezequiel)
+ * Ezequiel Rivero
+ * Patrick Mezard
+ * Rolando Espinoza
diff --git a/docs/experimental/crawlspider-v2.rst b/docs/experimental/crawlspider-v2.rst
new file mode 100644
index 000000000..a1ea09c48
--- /dev/null
+++ b/docs/experimental/crawlspider-v2.rst
@@ -0,0 +1,128 @@
+.. _topics-crawlspider-v2:
+
+==============
+CrawlSpider v2
+==============
+
+Introduction
+============
+
+TODO: introduction
+
+Rules Matching
+==============
+
+TODO: describe purpose of rules
+
+Request Extractors & Processors
+===============================
+
+TODO: describe purpose of extractors & processors
+
+Examples
+========
+
+TODO: plenty of examples
+
+
+.. module:: scrapy.contrib_exp.crawlspider.spider
+ :synopsis: CrawlSpider
+
+
+Reference
+=========
+
+CrawlSpider
+-----------
+
+TODO: describe crawlspider
+
+.. class:: CrawlSpider
+
+ TODO: describe class
+
+
+.. module:: scrapy.contrib_exp.crawlspider.rules
+ :synopsis: Rules
+
+Rules
+-----
+
+TODO: describe spider rules
+
+.. class:: Rule
+
+ TODO: describe Rules class
+
+
+.. module:: scrapy.contrib_exp.crawlspider.reqext
+ :synopsis: Request Extractors
+
+Request Extractors
+------------------
+
+TODO: describe extractors purpose
+
+.. class:: BaseSgmlRequestExtractor
+
+ TODO: describe base extractor
+
+.. class:: SgmlRequestExtractor
+
+ TODO: describe sgml extractor
+
+.. class:: XPathRequestExtractor
+
+ TODO: describe xpath request extractor
+
+
+.. module:: scrapy.contrib_exp.crawlspider.reqproc
+ :synopsis: Request Processors
+
+Request Processors
+------------------
+
+TODO: describe request processors
+
+.. class:: Canonicalize
+
+ TODO: describe proc
+
+.. class:: Unique
+
+ TODO: describe unique
+
+.. class:: FilterDomain
+
+ TODO: describe filter domain
+
+.. class:: FilterUrl
+
+ TODO: describe filter url
+
+
+.. module:: scrapy.contrib_exp.crawlspider.matchers
+ :synopsis: Matchers
+
+Request/Response Matchers
+-------------------------
+
+TODO: describe matchers
+
+.. class:: BaseMatcher
+
+ TODO: describe base matcher
+
+.. class:: UrlMatcher
+
+ TODO: describe url matcher
+
+.. class:: UrlRegexMatcher
+
+ TODO: describe UrlListMatcher
+
+.. class:: UrlListMatcher
+
+ TODO: describe url list matcher
+
+
diff --git a/docs/experimental/index.rst b/docs/experimental/index.rst
index 47f4ee4ac..f63ceddd9 100644
--- a/docs/experimental/index.rst
+++ b/docs/experimental/index.rst
@@ -21,3 +21,4 @@ it's properly merged) . Use at your own risk.
djangoitems
scheduler-middleware
+ crawlspider-v2
diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst
index e8aad451b..4358a7ae0 100644
--- a/docs/intro/tutorial.rst
+++ b/docs/intro/tutorial.rst
@@ -420,8 +420,8 @@ scraped so far, the code for our Spider should be like this::
Now doing a crawl on the dmoz.org domain yields ``DmozItem``'s::
- [dmoz.org] DEBUG: Scraped DmozItem({'title': [u'Text Processing in Python'], 'link': [u'http://gnosis.cx/TPiP/'], 'desc': [u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n']}) in
- [dmoz.org] DEBUG: Scraped DmozItem({'title': [u'XML Processing with Python'], 'link': [u'http://www.informit.com/store/product.aspx?isbn=0130211192'], 'desc': [u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n']}) in
+ [dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n'], link=[u'http://gnosis.cx/TPiP/'], title=[u'Text Processing in Python']) in
+ [dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n'], link=[u'http://www.informit.com/store/product.aspx?isbn=0130211192'], title=[u'XML Processing with Python']) in
Storing the data (using an Item Pipeline)
diff --git a/docs/topics/email.rst b/docs/topics/email.rst
index 7fc2f9d60..984132eaf 100644
--- a/docs/topics/email.rst
+++ b/docs/topics/email.rst
@@ -5,14 +5,17 @@ Sending email
=============
.. module:: scrapy.mail
- :synopsis: Helpers to easily send e-mail.
+ :synopsis: Email sending facility
Although Python makes sending e-mail relatively easy via the `smtplib`_
-library, Scrapy provides its own class for sending emails which is very easy to
-use and it's implemented using `Twisted non-blocking IO`_, to avoid affecting
-the crawling performance.
+library, Scrapy provides its own facility for sending emails which is very easy
+to use and it's implemented using `Twisted non-blocking IO`_, to avoid
+interfering with the non-blocking IO of the crawler.
+
+It's also very easy to configure by just configuring a few settings.
.. _smtplib: http://docs.python.org/library/smtplib.html
+.. _Twisted non-blocking IO: http://twistedmatrix.com/projects/core/documentation/howto/async.html
It also has built-in support for sending attachments.
@@ -34,30 +37,45 @@ uses `Twisted non-blocking IO`_, like the rest of the framework.
.. class:: MailSender(smtphost, mailfrom)
- ``smtphost`` is a string with the SMTP host to use for sending the emails.
- If omitted, :setting:`MAIL_HOST` will be used.
+ :param smtphost: the SMTP host to use for sending the emails. If omitted,
+ :setting:`MAIL_HOST` setting will be used.
+ :type smtphost: str
- ``mailfrom`` is a string with the email address to use for sending messages
- (in the ``From:`` header). If omitted, :setting:`MAIL_FROM` will be used.
+ :param mailfrom: the address used to send emails (in the ``From:`` header).
+ If omitted, :setting:`MAIL_FROM` setting will be used.
+ :type mailfrom: str
-.. method:: MailSender.send(to, subject, body, cc=None, attachs=())
+ .. method:: send(to, subject, body, cc=None, attachs=())
- Send mail to the given recipients
+ Send email to the given recipients
- ``to`` is a list of email recipients
+ :param to: the email recipients
+ :type to: list
- ``subject`` is a string with the subject of the message
+ :param subject: the subject of the email
+ :type subject: str
- ``cc`` is a list of emails to CC
+ :param cc: the emails to CC
+ :type cc: list
- ``body`` is a string with the body of the message
+ :param body: the email body
+ :type body: str
- ``attachs`` is an iterable of tuples (attach_name, mimetype, file_object)
- where:
-
- ``attach_name`` is a string with the name will appear on the emails attachment
- ``mimetype`` is the mimetype of the attachment
- ``file_object`` is a readable file object
+ :param attachs: an iterable of tuples ``(attach_name, mimetype,
+ file_object)`` where ``attach_name`` is a string with the name will
+ appear on the emails attachment, ``mimetype`` is the mimetype of the
+ attachment and ``file_object`` is a readable file object with the
+ contents of the attachment
+ :type attachs: iterable
-.. _Twisted non-blocking IO: http://twistedmatrix.com/projects/core/documentation/howto/async.html
+MailSender settings
+===================
+
+These settings define the default constructor values of the :class:`MailSender`
+class, and can be used to configure email notifications in your project without
+writing any code (for those extensions that use the :class:`MailSender` class):
+
+* :setting:`MAIL_FROM`
+* :setting:`MAIL_HOST`
+
diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst
index a72b285c3..019c9c32b 100644
--- a/docs/topics/item-pipeline.rst
+++ b/docs/topics/item-pipeline.rst
@@ -98,10 +98,10 @@ spider returns multiples items with the same id::
del self.duplicates[spider]
def process_item(self, spider, item):
- if item.id in self.duplicates[spider]:
+ if item['id'] in self.duplicates[spider]:
raise DropItem("Duplicate item found: %s" % item)
else:
- self.duplicates[spider].add(item.id)
+ self.duplicates[spider].add(item['id'])
return item
Built-in Item Pipelines reference
@@ -178,7 +178,7 @@ on the respective Item Exporter to get more info.
the first line. This format requires you to specify the fields to export
using the :setting:`EXPORT_FIELDS` setting.
-* ``jsonlines``: uses a :class:`~jsonlines.JsonLinesItemExporter`
+* ``json``: uses a :class:`~jsonlines.JsonLinesItemExporter`
* ``pickle``: uses a :class:`PickleItemExporter`
diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst
index f57128287..5e1a0e4bb 100644
--- a/docs/topics/logging.rst
+++ b/docs/topics/logging.rst
@@ -129,3 +129,14 @@ scrapy.log module
Log level for debugging messages (recommended level for development)
+Logging settings
+================
+
+These settings can be used to configure the logging:
+
+* :setting:`LOG_ENABLED`
+* :setting:`LOG_ENCODING`
+* :setting:`LOG_FILE`
+* :setting:`LOG_LEVEL`
+* :setting:`LOG_STDOUT`
+
diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst
index 3472b1937..63d289d09 100644
--- a/docs/topics/request-response.rst
+++ b/docs/topics/request-response.rst
@@ -466,12 +466,14 @@ TextResponse objects
.. attribute:: TextResponse.encoding
- A string with the encoding of this response. The encoding is resolved in the
- following order:
+ A string with the encoding of this response. The encoding is resolved by
+ trying the following mechanisms, in order:
1. the encoding passed in the constructor `encoding` argument
- 2. the encoding declared in the Content-Type HTTP header
+ 2. the encoding declared in the Content-Type HTTP header. If this
+ encoding is not valid (ie. unknown), it is ignored and the next
+ resolution mechanism is tried.
3. the encoding declared in the response body. The TextResponse class
doesn't provide any special functionality for this. However, the
@@ -483,23 +485,11 @@ TextResponse objects
:class:`TextResponse` objects support the following methods in addition to
the standard :class:`Response` ones:
- .. method:: TextResponse.headers_encoding()
-
- Returns a string with the encoding declared in the headers (ie. the
- Content-Type HTTP header).
-
- .. method:: TextResponse.body_encoding()
-
- Returns a string with the encoding of the body, either declared or inferred
- from its contents. The body encoding declaration is implemented in
- :class:`TextResponse` subclasses such as: :class:`HtmlResponse` or
- :class:`XmlResponse`.
-
.. method:: TextResponse.body_as_unicode()
Returns the body of the response as unicode. This is equivalent to::
- response.body.encode(response.encoding)
+ response.body.decode(response.encoding)
But **not** equivalent to::
diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst
index 4b3dbc78f..eacc2107c 100644
--- a/docs/topics/settings.rst
+++ b/docs/topics/settings.rst
@@ -340,16 +340,6 @@ Default: ``True``
Whether to collect depth stats.
-.. setting:: DOMAIN_SCHEDULER
-
-SPIDER_SCHEDULER
-----------------
-
-Default: ``'scrapy.contrib.spiderscheduler.FifoSpiderScheduler'``
-
-The Spider Scheduler to use. The spider scheduler returns the next spider to
-scrape.
-
.. setting:: DOWNLOADER_DEBUG
DOWNLOADER_DEBUG
@@ -418,6 +408,15 @@ supported. Example::
DOWNLOAD_DELAY = 0.25 # 250 ms of delay
+This setting is also affected by the :setting:`RANDOMIZE_DOWNLOAD_DELAY`
+setting (which is enabled by default). By default, Scrapy doesn't wait a fixed
+amount of time between requests, but uses a random interval between 0.5 and 1.5
+* :setting:`DOWNLOAD_DELAY`.
+
+Another way to change the download delay (per spider, instead of globally) is
+by using the ``download_delay`` spider attribute, which takes more precedence
+than this setting.
+
.. setting:: DOWNLOAD_TIMEOUT
DOWNLOAD_TIMEOUT
@@ -439,6 +438,52 @@ The class used to detect and filter duplicate requests.
The default (``RequestFingerprintDupeFilter``) filters based on request fingerprint
(using ``scrapy.utils.request.request_fingerprint``) and grouping per domain.
+.. setting:: ENCODING_ALIASES
+
+ENCODING_ALIASES
+----------------
+
+Default: ``{}``
+
+A mapping of custom encoding aliases for your project, where the keys are the
+aliases (and must be lower case) and the values are the encodings they map to.
+
+This setting extends the :setting:`ENCODING_ALIASES_BASE` setting which
+contains some default mappings.
+
+.. setting:: ENCODING_ALIASES_BASE
+
+ENCODING_ALIASES_BASE
+---------------------
+
+Default::
+
+ {
+ 'gb2312': 'zh-cn',
+ 'cp1251': 'win-1251',
+ 'macintosh' : 'mac-roman',
+ 'x-sjis': 'shift-jis',
+ 'iso-8859-1': 'cp1252',
+ 'iso8859-1': 'cp1252',
+ '8859': 'cp1252',
+ 'cp819': 'cp1252',
+ 'latin': 'cp1252',
+ 'latin1': 'cp1252',
+ 'latin_1': 'cp1252',
+ 'l1': 'cp1252',
+ }
+
+The default encoding aliases defined in Scrapy. Don't override this setting in
+your project, override :setting:`ENCODING_ALIASES` instead.
+
+The reason why `ISO-8859-1`_ (and all its aliases) are mapped to `CP1252`_ is
+due to a well known browser hack. For more information see: `Character
+encodings in HTML`_.
+
+.. _ISO-8859-1: http://en.wikipedia.org/wiki/ISO/IEC_8859-1
+.. _CP1252: http://en.wikipedia.org/wiki/Windows-1252
+.. _Character encodings in HTML: http://en.wikipedia.org/wiki/Character_encodings_in_HTML
+
.. setting:: EXTENSIONS
EXTENSIONS
@@ -517,7 +562,16 @@ LOG_ENABLED
Default: ``True``
-Enable logging.
+Whether to enable logging.
+
+.. setting:: LOG_ENCODING
+
+LOG_ENCODING
+------------
+
+Default: ``'utf-8'``
+
+The encoding to use for logging.
.. setting:: LOG_FILE
@@ -677,6 +731,27 @@ Example::
NEWSPIDER_MODULE = 'mybot.spiders_dev'
+.. setting:: RANDOMIZE_DOWNLOAD_DELAY
+
+RANDOMIZE_DOWNLOAD_DELAY
+------------------------
+
+Default: ``True``
+
+If enabled, Scrapy will wait a random amount of time (between 0.5 and 1.5
+* :setting:`DOWNLOAD_DELAY`) while fetching requests from the same
+spider.
+
+This randomization decreases the chance of the crawler being detected (and
+subsequently blocked) by sites which analyze requests looking for statistically
+significant similarities in the time between their times.
+
+The randomization policy is the same used by `wget`_ ``--random-wait`` option.
+
+If :setting:`DOWNLOAD_DELAY` is zero (default) this option has no effect.
+
+.. _wget: http://www.gnu.org/software/wget/manual/wget.html
+
.. setting:: REDIRECT_MAX_TIMES
REDIRECT_MAX_TIMES
@@ -773,7 +848,7 @@ The scheduler to use for crawling.
SCHEDULER_ORDER
---------------
-Default: ``'BFO'``
+Default: ``'DFO'``
Scope: ``scrapy.core.scheduler``
@@ -858,6 +933,16 @@ Example::
SPIDER_MODULES = ['mybot.spiders_prod', 'mybot.spiders_dev']
+.. setting:: SPIDER_SCHEDULER
+
+SPIDER_SCHEDULER
+----------------
+
+Default: ``'scrapy.contrib.spiderscheduler.FifoSpiderScheduler'``
+
+The Spider Scheduler to use. The spider scheduler returns the next spider to
+scrape.
+
.. setting:: STATS_CLASS
STATS_CLASS
diff --git a/examples/experimental/googledir/googledir/__init__.py b/examples/experimental/googledir/googledir/__init__.py
new file mode 100644
index 000000000..3104ef709
--- /dev/null
+++ b/examples/experimental/googledir/googledir/__init__.py
@@ -0,0 +1 @@
+# googledir project
diff --git a/examples/experimental/googledir/googledir/items.py b/examples/experimental/googledir/googledir/items.py
new file mode 100644
index 000000000..decc2c9ba
--- /dev/null
+++ b/examples/experimental/googledir/googledir/items.py
@@ -0,0 +1,16 @@
+# Define here the models for your scraped items
+#
+# See documentation in:
+# http://doc.scrapy.org/topics/items.html
+
+from scrapy.item import Item, Field
+
+class GoogledirItem(Item):
+
+ name = Field(default='')
+ url = Field(default='')
+ description = Field(default='')
+
+ def __str__(self):
+ return "Google Category: name=%s url=%s" \
+ % (self['name'], self['url'])
diff --git a/examples/experimental/googledir/googledir/pipelines.py b/examples/experimental/googledir/googledir/pipelines.py
new file mode 100644
index 000000000..f775b254c
--- /dev/null
+++ b/examples/experimental/googledir/googledir/pipelines.py
@@ -0,0 +1,22 @@
+# Define your item pipelines here
+#
+# Don't forget to add your pipeline to the ITEM_PIPELINES setting
+# See: http://doc.scrapy.org/topics/item-pipeline.html
+
+from scrapy.core.exceptions import DropItem
+
+class FilterWordsPipeline(object):
+ """
+ A pipeline for filtering out items which contain certain
+ words in their description
+ """
+
+ # put all words in lowercase
+ words_to_filter = ['politics', 'religion']
+
+ def process_item(self, spider, item):
+ for word in self.words_to_filter:
+ if word in unicode(item['description']).lower():
+ raise DropItem("Contains forbidden word: %s" % word)
+ else:
+ return item
diff --git a/examples/experimental/googledir/googledir/settings.py b/examples/experimental/googledir/googledir/settings.py
new file mode 100644
index 000000000..4e3c11163
--- /dev/null
+++ b/examples/experimental/googledir/googledir/settings.py
@@ -0,0 +1,21 @@
+# Scrapy settings for googledir project
+#
+# For simplicity, this file contains only the most important settings by
+# default. All the other settings are documented here:
+#
+# http://doc.scrapy.org/topics/settings.html
+#
+# Or you can copy and paste them from where they're defined in Scrapy:
+#
+# scrapy/conf/default_settings.py
+#
+
+BOT_NAME = 'googledir'
+BOT_VERSION = '1.0'
+
+SPIDER_MODULES = ['googledir.spiders']
+NEWSPIDER_MODULE = 'googledir.spiders'
+DEFAULT_ITEM_CLASS = 'googledir.items.GoogledirItem'
+USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION)
+
+ITEM_PIPELINES = ['googledir.pipelines.FilterWordsPipeline']
diff --git a/examples/experimental/googledir/googledir/spiders/__init__.py b/examples/experimental/googledir/googledir/spiders/__init__.py
new file mode 100644
index 000000000..5065ccba5
--- /dev/null
+++ b/examples/experimental/googledir/googledir/spiders/__init__.py
@@ -0,0 +1,8 @@
+# This package will contain the spiders of your Scrapy project
+#
+# To create the first spider for your project use this command:
+#
+# scrapy-ctl.py genspider myspider myspider-domain.com
+#
+# For more info see:
+# http://doc.scrapy.org/topics/spiders.html
diff --git a/examples/experimental/googledir/googledir/spiders/google_directory.py b/examples/experimental/googledir/googledir/spiders/google_directory.py
new file mode 100644
index 000000000..fdbf3f24a
--- /dev/null
+++ b/examples/experimental/googledir/googledir/spiders/google_directory.py
@@ -0,0 +1,40 @@
+from scrapy.selector import HtmlXPathSelector
+from scrapy.contrib.loader import XPathItemLoader
+from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule
+
+from googledir.items import GoogledirItem
+
+class GoogleDirectorySpider(CrawlSpider):
+
+ domain_name = 'directory.google.com'
+ start_urls = ['http://directory.google.com/']
+
+ rules = (
+ # search for categories pattern and follow links
+ Rule(r'/[A-Z][a-zA-Z_/]+$', 'parse_category', follow=True),
+ )
+
+ def parse_category(self, response):
+ # The main selector we're using to extract data from the page
+ main_selector = HtmlXPathSelector(response)
+
+ # The XPath to website links in the directory page
+ xpath = '//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td/font'
+
+ # Get a list of (sub) selectors to each website node pointed by the XPath
+ sub_selectors = main_selector.select(xpath)
+
+ # Iterate over the sub-selectors to extract data for each website
+ for selector in sub_selectors:
+ item = GoogledirItem()
+
+ l = XPathItemLoader(item=item, selector=selector)
+ l.add_xpath('name', 'a/text()')
+ l.add_xpath('url', 'a/@href')
+ l.add_xpath('description', 'font[2]/text()')
+
+ # Here we populate the item and yield it
+ yield l.load_item()
+
+SPIDER = GoogleDirectorySpider()
+
diff --git a/examples/experimental/googledir/scrapy-ctl.py b/examples/experimental/googledir/scrapy-ctl.py
new file mode 100644
index 000000000..552421ac3
--- /dev/null
+++ b/examples/experimental/googledir/scrapy-ctl.py
@@ -0,0 +1,7 @@
+#!/usr/bin/env python
+
+import os
+os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'googledir.settings')
+
+from scrapy.command.cmdline import execute
+execute()
diff --git a/examples/experimental/imdb/imdb/__init__.py b/examples/experimental/imdb/imdb/__init__.py
new file mode 100644
index 000000000..5bb534f79
--- /dev/null
+++ b/examples/experimental/imdb/imdb/__init__.py
@@ -0,0 +1 @@
+# package
diff --git a/examples/experimental/imdb/imdb/items.py b/examples/experimental/imdb/imdb/items.py
new file mode 100644
index 000000000..03bb5c2c3
--- /dev/null
+++ b/examples/experimental/imdb/imdb/items.py
@@ -0,0 +1,12 @@
+# Define here the models for your scraped items
+#
+# See documentation in:
+# http://doc.scrapy.org/topics/items.html
+
+from scrapy.item import Item, Field
+
+class ImdbItem(Item):
+ # define the fields for your item here like:
+ # name = Field()
+ title = Field()
+ url = Field()
diff --git a/examples/experimental/imdb/imdb/pipelines.py b/examples/experimental/imdb/imdb/pipelines.py
new file mode 100644
index 000000000..e60714159
--- /dev/null
+++ b/examples/experimental/imdb/imdb/pipelines.py
@@ -0,0 +1,8 @@
+# Define your item pipelines here
+#
+# Don't forget to add your pipeline to the ITEM_PIPELINES setting
+# See: http://doc.scrapy.org/topics/item-pipeline.html
+
+class ImdbPipeline(object):
+ def process_item(self, spider, item):
+ return item
diff --git a/examples/experimental/imdb/imdb/settings.py b/examples/experimental/imdb/imdb/settings.py
new file mode 100644
index 000000000..de026dc14
--- /dev/null
+++ b/examples/experimental/imdb/imdb/settings.py
@@ -0,0 +1,20 @@
+# Scrapy settings for imdb project
+#
+# For simplicity, this file contains only the most important settings by
+# default. All the other settings are documented here:
+#
+# http://doc.scrapy.org/topics/settings.html
+#
+# Or you can copy and paste them from where they're defined in Scrapy:
+#
+# scrapy/conf/default_settings.py
+#
+
+BOT_NAME = 'imdb'
+BOT_VERSION = '1.0'
+
+SPIDER_MODULES = ['imdb.spiders']
+NEWSPIDER_MODULE = 'imdb.spiders'
+DEFAULT_ITEM_CLASS = 'imdb.items.ImdbItem'
+USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION)
+
diff --git a/examples/experimental/imdb/imdb/spiders/__init__.py b/examples/experimental/imdb/imdb/spiders/__init__.py
new file mode 100644
index 000000000..5065ccba5
--- /dev/null
+++ b/examples/experimental/imdb/imdb/spiders/__init__.py
@@ -0,0 +1,8 @@
+# This package will contain the spiders of your Scrapy project
+#
+# To create the first spider for your project use this command:
+#
+# scrapy-ctl.py genspider myspider myspider-domain.com
+#
+# For more info see:
+# http://doc.scrapy.org/topics/spiders.html
diff --git a/examples/experimental/imdb/imdb/spiders/imdb_site.py b/examples/experimental/imdb/imdb/spiders/imdb_site.py
new file mode 100644
index 000000000..5c3f3c06b
--- /dev/null
+++ b/examples/experimental/imdb/imdb/spiders/imdb_site.py
@@ -0,0 +1,140 @@
+from scrapy.http import Request
+from scrapy.selector import HtmlXPathSelector
+from scrapy.contrib.loader import XPathItemLoader
+from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize, \
+ FilterDupes, FilterUrl
+from scrapy.utils.url import urljoin_rfc
+
+from imdb.items import ImdbItem, Field
+
+from itertools import chain, imap, izip
+
+class UsaOpeningWeekMovie(ImdbItem):
+ pass
+
+class UsaTopWeekMovie(ImdbItem):
+ pass
+
+class Top250Movie(ImdbItem):
+ rank = Field()
+ rating = Field()
+ year = Field()
+ votes = Field()
+
+class MovieItem(ImdbItem):
+ release_date = Field()
+ tagline = Field()
+
+
+class ImdbSiteSpider(CrawlSpider):
+ domain_name = 'imdb.com'
+ start_urls = ['http://www.imdb.com/']
+
+ # extract requests using this classes from urls matching 'follow' flag
+ request_extractors = [
+ SgmlRequestExtractor(tags=['a'], attrs=['href']),
+ ]
+
+ # process requests using this classes from urls matching 'follow' flag
+ request_processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterUrl(deny=r'/tt\d+/$'), # deny movie url as we will dispatch
+ # manually the movie requests
+ ]
+
+ # include domain bit for demo purposes
+ rules = (
+ # these two rules expects requests from start url
+ Rule(r'imdb.com/nowplaying/$', 'parse_now_playing'),
+ Rule(r'imdb.com/chart/top$', 'parse_top_250'),
+ # this rule will parse requests manually dispatched
+ Rule(r'imdb.com/title/tt\d+/$', 'parse_movie_info'),
+ )
+
+ def parse_now_playing(self, response):
+ """Scrapes USA openings this week and top 10 in week"""
+ self.log("Parsing USA Top Week")
+ hxs = HtmlXPathSelector(response)
+
+ _urljoin = lambda url: self._urljoin(response, url)
+
+ #
+ # openings this week
+ #
+ openings = hxs.select('//table[@class="movies"]//a[@class="title"]')
+ boxoffice = hxs.select('//table[@class="boxoffice movies"]//a[@class="title"]')
+
+ opening_titles = openings.select('text()').extract()
+ opening_urls = imap(_urljoin, openings.select('@href').extract())
+
+ box_titles = boxoffice.select('text()').extract()
+ box_urls = imap(_urljoin, boxoffice.select('@href').extract())
+
+ # items
+ opening_items = (UsaOpeningWeekMovie(title=title, url=url)
+ for (title, url)
+ in izip(opening_titles, opening_urls))
+
+ box_items = (UsaTopWeekMovie(title=title, url=url)
+ for (title, url)
+ in izip(box_titles, box_urls))
+
+ # movie requests
+ requests = imap(self.make_requests_from_url,
+ chain(opening_urls, box_urls))
+
+ return chain(opening_items, box_items, requests)
+
+ def parse_top_250(self, response):
+ """Scrapes movies from top 250 list"""
+ self.log("Parsing Top 250")
+ hxs = HtmlXPathSelector(response)
+
+ # scrap each row in the table
+ rows = hxs.select('//div[@id="main"]/table/tr//a/ancestor::tr')
+ for row in rows:
+ fields = row.select('td//text()').extract()
+ url, = row.select('td//a/@href').extract()
+ url = self._urljoin(response, url)
+
+ item = Top250Movie()
+ item['title'] = fields[2]
+ item['url'] = url
+ item['rank'] = fields[0]
+ item['rating'] = fields[1]
+ item['year'] = fields[3]
+ item['votes'] = fields[4]
+
+ # scrapped top250 item
+ yield item
+ # fetch movie
+ yield self.make_requests_from_url(url)
+
+ def parse_movie_info(self, response):
+ """Scrapes movie information"""
+ self.log("Parsing Movie Info")
+ hxs = HtmlXPathSelector(response)
+ selector = hxs.select('//div[@class="maindetails"]')
+
+ item = MovieItem()
+ # set url
+ item['url'] = response.url
+
+ # use item loader for other attributes
+ l = XPathItemLoader(item=item, selector=selector)
+ l.add_xpath('title', './/h1/text()')
+ l.add_xpath('release_date', './/h5[text()="Release Date:"]'
+ '/following-sibling::div/text()')
+ l.add_xpath('tagline', './/h5[text()="Tagline:"]'
+ '/following-sibling::div/text()')
+
+ yield l.load_item()
+
+ def _urljoin(self, response, url):
+ """Helper to convert relative urls to absolute"""
+ return urljoin_rfc(response.url, url, response.encoding)
+
+SPIDER = ImdbSiteSpider()
diff --git a/examples/experimental/imdb/scrapy-ctl.py b/examples/experimental/imdb/scrapy-ctl.py
new file mode 100644
index 000000000..df57621b3
--- /dev/null
+++ b/examples/experimental/imdb/scrapy-ctl.py
@@ -0,0 +1,7 @@
+#!/usr/bin/env python
+
+import os
+os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'imdb.settings')
+
+from scrapy.command.cmdline import execute
+execute()
diff --git a/examples/scripts/count_and_follow_links.py b/examples/scripts/count_and_follow_links.py
deleted file mode 100644
index 4ead870fc..000000000
--- a/examples/scripts/count_and_follow_links.py
+++ /dev/null
@@ -1,51 +0,0 @@
-"""
-Simple script to follow links from a start url. The links are followed in no
-particular order.
-
-Usage:
-count_and_follow_links.py
-
-Example:
-count_and_follow_links.py http://scrapy.org/ 20
-
-For each page visisted, this script will print the page body size and the
-number of links found.
-"""
-
-import sys
-from urlparse import urljoin
-
-from scrapy.crawler import Crawler
-from scrapy.selector import HtmlXPathSelector
-from scrapy.http import Request, HtmlResponse
-
-links_followed = 0
-
-def parse(response):
- global links_followed
- links_followed += 1
- if links_followed >= links_to_follow:
- crawler.stop()
-
- # ignore non-HTML responses
- if not isinstance(response, HtmlResponse):
- return
-
- links = HtmlXPathSelector(response).select('//a/@href').extract()
- abslinks = [urljoin(response.url, l) for l in links]
-
- print "page %2d/%d: %s" % (links_followed, links_to_follow, response.url)
- print " size : %d bytes" % len(response.body)
- print " links: %d" % len(links)
- print
-
- return [Request(l, callback=parse) for l in abslinks]
-
-if len(sys.argv) != 3:
- print __doc__
- sys.exit(2)
-
-start_url, links_to_follow = sys.argv[1], int(sys.argv[2])
-request = Request(start_url, callback=parse)
-crawler = Crawler()
-crawler.crawl(request)
diff --git a/scrapy/__init__.py b/scrapy/__init__.py
index 2ecc7749c..184b51cb1 100644
--- a/scrapy/__init__.py
+++ b/scrapy/__init__.py
@@ -2,8 +2,8 @@
Scrapy - a screen scraping framework written in Python
"""
-version_info = (0, 8, 0, '', 0)
-__version__ = "0.8"
+version_info = (0, 9, 0, 'dev')
+__version__ = "0.9-dev"
import sys, os, warnings
@@ -17,11 +17,6 @@ warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
# monkey patches to fix external library issues
from scrapy.xlib import twisted_250_monkeypatches
-# add some common encoding aliases not included by default in Python
-from scrapy.utils.encoding import add_encoding_alias
-add_encoding_alias('gb2312', 'zh-cn')
-add_encoding_alias('cp1251', 'win-1251')
-
# optional_features is a set containing Scrapy optional features
optional_features = set()
diff --git a/scrapy/command/cmdline.py b/scrapy/command/cmdline.py
index 25f531cea..1a190549d 100644
--- a/scrapy/command/cmdline.py
+++ b/scrapy/command/cmdline.py
@@ -7,20 +7,14 @@ import cProfile
import scrapy
from scrapy import log
-from scrapy.spider import spiders
from scrapy.xlib import lsprofcalltree
from scrapy.conf import settings
from scrapy.command.models import ScrapyCommand
+from scrapy.utils.signal import send_catch_log
-# This dict holds information about the executed command for later use
-command_executed = {}
-
-def _save_command_executed(cmdname, cmd, args, opts):
- """Save command executed info for later reference"""
- command_executed['name'] = cmdname
- command_executed['class'] = cmd
- command_executed['args'] = args[:]
- command_executed['opts'] = opts.__dict__.copy()
+# Signal that carries information about the command which was executed
+# args: cmdname, cmdobj, args, opts
+command_executed = object()
def _find_commands(dir):
try:
@@ -127,7 +121,8 @@ def execute(argv=None):
sys.exit(2)
del args[0] # remove command name from args
- _save_command_executed(cmdname, cmd, args, opts)
+ send_catch_log(signal=command_executed, cmdname=cmdname, cmdobj=cmd, \
+ args=args, opts=opts)
from scrapy.core.manager import scrapymanager
scrapymanager.configure(control_reactor=True)
ret = _run_command(cmd, args, opts)
@@ -136,23 +131,25 @@ def execute(argv=None):
def _run_command(cmd, args, opts):
if opts.profile or opts.lsprof:
- if opts.profile:
- log.msg("writing cProfile stats to %r" % opts.profile)
- if opts.lsprof:
- log.msg("writing lsprof stats to %r" % opts.lsprof)
- loc = locals()
- p = cProfile.Profile()
- p.runctx('ret = cmd.run(args, opts)', globals(), loc)
- if opts.profile:
- p.dump_stats(opts.profile)
- k = lsprofcalltree.KCacheGrind(p)
- if opts.lsprof:
- with open(opts.lsprof, 'w') as f:
- k.output(f)
- ret = loc['ret']
+ return _run_command_profiled(cmd, args, opts)
else:
- ret = cmd.run(args, opts)
- return ret
+ return cmd.run(args, opts)
+
+def _run_command_profiled(cmd, args, opts):
+ if opts.profile:
+ log.msg("writing cProfile stats to %r" % opts.profile)
+ if opts.lsprof:
+ log.msg("writing lsprof stats to %r" % opts.lsprof)
+ loc = locals()
+ p = cProfile.Profile()
+ p.runctx('ret = cmd.run(args, opts)', globals(), loc)
+ if opts.profile:
+ p.dump_stats(opts.profile)
+ k = lsprofcalltree.KCacheGrind(p)
+ if opts.lsprof:
+ with open(opts.lsprof, 'w') as f:
+ k.output(f)
+ return loc['ret']
if __name__ == '__main__':
execute()
diff --git a/scrapy/conf/default_settings.py b/scrapy/conf/default_settings.py
index 8892e41b4..5e7bea3cd 100644
--- a/scrapy/conf/default_settings.py
+++ b/scrapy/conf/default_settings.py
@@ -71,6 +71,23 @@ DOWNLOADER_STATS = True
DUPEFILTER_CLASS = 'scrapy.contrib.dupefilter.RequestFingerprintDupeFilter'
+ENCODING_ALIASES = {}
+
+ENCODING_ALIASES_BASE = {
+ 'zh-cn': 'gb2312',
+ 'win-1251': 'cp1251',
+ 'macintosh' : 'mac-roman',
+ 'x-sjis': 'shift-jis',
+ 'iso-8859-1': 'cp1252',
+ 'iso8859-1': 'cp1252',
+ '8859': 'cp1252',
+ 'cp819': 'cp1252',
+ 'latin': 'cp1252',
+ 'latin1': 'cp1252',
+ 'latin_1': 'cp1252',
+ 'l1': 'cp1252',
+}
+
EXTENSIONS = {}
EXTENSIONS_BASE = {
@@ -101,6 +118,7 @@ ITEM_PROCESSOR = 'scrapy.contrib.pipeline.ItemPipelineManager'
ITEM_PIPELINES = []
LOG_ENABLED = True
+LOG_ENCODING = 'utf-8'
LOG_FORMATTER_CRAWLED = 'scrapy.contrib.logformatter.crawled_logline'
LOG_STDOUT = False
LOG_LEVEL = 'DEBUG'
@@ -122,6 +140,8 @@ MYSQL_CONNECTION_SETTINGS = {}
NEWSPIDER_MODULE = ''
+RANDOMIZE_DOWNLOAD_DELAY = True
+
REDIRECT_MAX_METAREFRESH_DELAY = 100
REDIRECT_MAX_TIMES = 20 # uses Firefox default setting
REDIRECT_PRIORITY_ADJUST = +2
@@ -150,7 +170,7 @@ SCHEDULER_MIDDLEWARES_BASE = {
'scrapy.contrib.schedulermiddleware.duplicatesfilter.DuplicatesFilterMiddleware': 500,
}
-SCHEDULER_ORDER = 'BFO' # available orders: BFO (default), DFO
+SCHEDULER_ORDER = 'DFO'
SPIDER_MANAGER_CLASS = 'scrapy.contrib.spidermanager.TwistedPluginSpiderManager'
diff --git a/scrapy/contrib/downloadermiddleware/redirect.py b/scrapy/contrib/downloadermiddleware/redirect.py
index 249862824..1c6297b49 100644
--- a/scrapy/contrib/downloadermiddleware/redirect.py
+++ b/scrapy/contrib/downloadermiddleware/redirect.py
@@ -1,4 +1,5 @@
from scrapy import log
+from scrapy.http import HtmlResponse
from scrapy.utils.url import urljoin_rfc
from scrapy.utils.response import get_meta_refresh
from scrapy.core.exceptions import IgnoreRequest
@@ -24,10 +25,11 @@ class RedirectMiddleware(object):
redirected = request.replace(url=redirected_url)
return self._redirect(redirected, request, spider, response.status)
- interval, url = get_meta_refresh(response)
- if url and interval < self.max_metarefresh_delay:
- redirected = self._redirect_request_using_get(request, url)
- return self._redirect(redirected, request, spider, 'meta refresh')
+ if isinstance(response, HtmlResponse):
+ interval, url = get_meta_refresh(response)
+ if url and interval < self.max_metarefresh_delay:
+ redirected = self._redirect_request_using_get(request, url)
+ return self._redirect(redirected, request, spider, 'meta refresh')
return response
diff --git a/scrapy/contrib/groupsettings.py b/scrapy/contrib/groupsettings.py
deleted file mode 100644
index 4bad4100b..000000000
--- a/scrapy/contrib/groupsettings.py
+++ /dev/null
@@ -1,26 +0,0 @@
-"""
-Extensions to override scrapy settings with per-group settings according to the
-group the spider belongs to. It only overrides the settings when running the
-crawl command with *only one domain as argument*.
-"""
-
-from scrapy.conf import settings
-from scrapy.core.exceptions import NotConfigured
-from scrapy.command.cmdline import command_executed
-
-class GroupSettings(object):
-
- def __init__(self):
- if not settings.getbool("GROUPSETTINGS_ENABLED"):
- raise NotConfigured
-
- if command_executed and command_executed['name'] == 'crawl':
- mod = __import__(settings['GROUPSETTINGS_MODULE'], {}, {}, [''])
- args = command_executed['args']
- if len(args) == 1 and not args[0].startswith('http://'):
- domain = args[0]
- settings.overrides.update(mod.default_settings)
- for group, domains in mod.group_spiders.iteritems():
- if domain in domains:
- settings.overrides.update(mod.group_settings.get(group, {}))
-
diff --git a/scrapy/contrib/linkextractors/htmlparser.py b/scrapy/contrib/linkextractors/htmlparser.py
index 2714fb562..fb3fd661b 100644
--- a/scrapy/contrib/linkextractors/htmlparser.py
+++ b/scrapy/contrib/linkextractors/htmlparser.py
@@ -26,7 +26,7 @@ class HtmlParserLinkExtractor(HTMLParser):
links = unique_list(self.links, key=lambda link: link.url) if self.unique else self.links
ret = []
- base_url = self.base_url if self.base_url else response_url
+ base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
for link in links:
link.url = urljoin_rfc(base_url, link.url, response_encoding)
link.url = safe_url_string(link.url, response_encoding)
diff --git a/scrapy/contrib/linkextractors/image.py b/scrapy/contrib/linkextractors/image.py
index 711623b81..88dd78577 100644
--- a/scrapy/contrib/linkextractors/image.py
+++ b/scrapy/contrib/linkextractors/image.py
@@ -3,7 +3,6 @@ This module implements the HtmlImageLinkExtractor for extracting
image links only.
"""
-import urlparse
from scrapy.link import Link
from scrapy.utils.url import canonicalize_url, urljoin_rfc
@@ -25,13 +24,13 @@ class HTMLImageLinkExtractor(object):
self.unique = unique
self.canonicalize = canonicalize
- def extract_from_selector(self, selector, parent=None):
+ def extract_from_selector(self, selector, encoding, parent=None):
ret = []
def _add_link(url_sel, alt_sel=None):
url = flatten([url_sel.extract()])
alt = flatten([alt_sel.extract()]) if alt_sel else (u'', )
if url:
- ret.append(Link(unicode_to_str(url[0]), alt[0]))
+ ret.append(Link(unicode_to_str(url[0], encoding), alt[0]))
if selector.xmlNode.type == 'element':
if selector.xmlNode.name == 'img':
@@ -41,7 +40,7 @@ class HTMLImageLinkExtractor(object):
children = selector.select('child::*')
if len(children):
for child in children:
- ret.extend(self.extract_from_selector(child, parent=selector))
+ ret.extend(self.extract_from_selector(child, encoding, parent=selector))
elif selector.xmlNode.name == 'a' and not parent:
_add_link(selector.select('@href'), selector.select('@title'))
else:
@@ -52,7 +51,7 @@ class HTMLImageLinkExtractor(object):
def extract_links(self, response):
xs = HtmlXPathSelector(response)
base_url = xs.select('//base/@href').extract()
- base_url = unicode_to_str(base_url[0]) if base_url else unicode_to_str(response.url)
+ base_url = urljoin_rfc(response.url, base_url[0]) if base_url else response.url
links = []
for location in self.locations:
@@ -64,7 +63,7 @@ class HTMLImageLinkExtractor(object):
continue
for selector in selectors:
- links.extend(self.extract_from_selector(selector))
+ links.extend(self.extract_from_selector(selector, response.encoding))
seen, ret = set(), []
for link in links:
diff --git a/scrapy/contrib/linkextractors/lxmlparser.py b/scrapy/contrib/linkextractors/lxmlparser.py
index 390e3a304..27cd0697a 100644
--- a/scrapy/contrib/linkextractors/lxmlparser.py
+++ b/scrapy/contrib/linkextractors/lxmlparser.py
@@ -29,7 +29,7 @@ class LxmlLinkExtractor(object):
links = unique_list(self.links, key=lambda link: link.url) if self.unique else self.links
ret = []
- base_url = self.base_url if self.base_url else response_url
+ base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
for link in links:
link.url = urljoin_rfc(base_url, link.url, response_encoding)
link.url = safe_url_string(link.url, response_encoding)
diff --git a/scrapy/contrib/linkextractors/regex.py b/scrapy/contrib/linkextractors/regex.py
index 08e1e4526..1de044df3 100644
--- a/scrapy/contrib/linkextractors/regex.py
+++ b/scrapy/contrib/linkextractors/regex.py
@@ -16,8 +16,9 @@ def clean_link(link_text):
class RegexLinkExtractor(SgmlLinkExtractor):
"""High performant link extractor"""
+
def _extract_links(self, response_text, response_url, response_encoding):
- base_url = self.base_url if self.base_url else response_url
+ base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
clean_url = lambda u: urljoin_rfc(base_url, remove_entities(clean_link(u.decode(response_encoding))))
clean_text = lambda t: replace_escape_chars(remove_tags(t.decode(response_encoding))).strip()
diff --git a/scrapy/contrib/linkextractors/sgml.py b/scrapy/contrib/linkextractors/sgml.py
index d548626ab..9ec664bda 100644
--- a/scrapy/contrib/linkextractors/sgml.py
+++ b/scrapy/contrib/linkextractors/sgml.py
@@ -28,7 +28,7 @@ class BaseSgmlLinkExtractor(FixedSGMLParser):
links = unique_list(self.links, key=lambda link: link.url) if self.unique else self.links
ret = []
- base_url = self.base_url if self.base_url else response_url
+ base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
for link in links:
link.url = urljoin_rfc(base_url, link.url, response_encoding)
link.url = safe_url_string(link.url, response_encoding)
diff --git a/scrapy/contrib_exp/crawlspider/__init__.py b/scrapy/contrib_exp/crawlspider/__init__.py
new file mode 100644
index 000000000..03173eb38
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/__init__.py
@@ -0,0 +1,4 @@
+"""CrawlSpider v2"""
+
+from .rules import Rule
+from .spider import CrawlSpider
diff --git a/scrapy/contrib_exp/crawlspider/matchers.py b/scrapy/contrib_exp/crawlspider/matchers.py
new file mode 100644
index 000000000..3ef259c67
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/matchers.py
@@ -0,0 +1,61 @@
+"""
+Request/Response Matchers
+
+Perform evaluation to Request or Response attributes
+"""
+
+import re
+
+class BaseMatcher(object):
+ """Base matcher. Returns True by default."""
+
+ def matches_request(self, request):
+ """Performs Request Matching"""
+ return True
+
+ def matches_response(self, response):
+ """Performs Response Matching"""
+ return True
+
+
+class UrlMatcher(BaseMatcher):
+ """Matches URL attribute"""
+
+ def __init__(self, url):
+ """Initialize url attribute"""
+ self._url = url
+
+ def matches_url(self, url):
+ """Returns True if given url is equal to matcher's url"""
+ return self._url == url
+
+ def matches_request(self, request):
+ """Returns True if Request's url matches initial url"""
+ return self.matches_url(request.url)
+
+ def matches_response(self, response):
+ """Returns True if Response's url matches initial url"""
+ return self.matches_url(response.url)
+
+
+class UrlRegexMatcher(UrlMatcher):
+ """Matches URL using regular expression"""
+
+ def __init__(self, regex, flags=0):
+ """Initialize regular expression"""
+ self._regex = re.compile(regex, flags)
+
+ def matches_url(self, url):
+ """Returns True if url matches regular expression"""
+ return self._regex.search(url) is not None
+
+
+class UrlListMatcher(UrlMatcher):
+ """Matches if URL is in List"""
+
+ def __init__(self, urls):
+ self._urls = urls
+
+ def matches_url(self, url):
+ """Returns True if url is in urls list"""
+ return url in self._urls
diff --git a/scrapy/contrib_exp/crawlspider/reqext.py b/scrapy/contrib_exp/crawlspider/reqext.py
new file mode 100644
index 000000000..e23e78082
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/reqext.py
@@ -0,0 +1,117 @@
+"""Request Extractors"""
+from scrapy.http import Request
+from scrapy.selector import HtmlXPathSelector
+from scrapy.utils.misc import arg_to_iter
+from scrapy.utils.python import FixedSGMLParser, str_to_unicode
+from scrapy.utils.url import safe_url_string, urljoin_rfc
+
+from itertools import ifilter
+
+
+class BaseSgmlRequestExtractor(FixedSGMLParser):
+ """Base SGML Request Extractor"""
+
+ def __init__(self, tag='a', attr='href'):
+ """Initialize attributes"""
+ FixedSGMLParser.__init__(self)
+
+ self.scan_tag = tag if callable(tag) else lambda t: t == tag
+ self.scan_attr = attr if callable(attr) else lambda a: a == attr
+ self.current_request = None
+
+ def extract_requests(self, response):
+ """Returns list of requests extracted from response"""
+ return self._extract_requests(response.body, response.url,
+ response.encoding)
+
+ def _extract_requests(self, response_text, response_url, response_encoding):
+ """Extract requests with absolute urls"""
+ self.reset()
+ self.feed(response_text)
+ self.close()
+
+ base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
+ self._make_absolute_urls(base_url, response_encoding)
+ self._fix_link_text_encoding(response_encoding)
+
+ return self.requests
+
+ def _make_absolute_urls(self, base_url, encoding):
+ """Makes all request's urls absolute"""
+ for req in self.requests:
+ url = req.url
+ # make absolute url
+ url = urljoin_rfc(base_url, url, encoding)
+ url = safe_url_string(url, encoding)
+ # replace in-place request's url
+ req.url = url
+
+ def _fix_link_text_encoding(self, encoding):
+ """Convert link_text to unicode for each request"""
+ for req in self.requests:
+ req.meta.setdefault('link_text', '')
+ req.meta['link_text'] = str_to_unicode(req.meta['link_text'],
+ encoding)
+
+ def reset(self):
+ """Reset state"""
+ FixedSGMLParser.reset(self)
+ self.requests = []
+ self.base_url = None
+
+ def unknown_starttag(self, tag, attrs):
+ """Process unknown start tag"""
+ if 'base' == tag:
+ self.base_url = dict(attrs).get('href')
+
+ _matches = lambda (attr, value): self.scan_attr(attr) \
+ and value is not None
+ if self.scan_tag(tag):
+ for attr, value in ifilter(_matches, attrs):
+ req = Request(url=value)
+ self.requests.append(req)
+ self.current_request = req
+
+ def unknown_endtag(self, tag):
+ """Process unknown end tag"""
+ self.current_request = None
+
+ def handle_data(self, data):
+ """Process data"""
+ current = self.current_request
+ if current and not 'link_text' in current.meta:
+ current.meta['link_text'] = data.strip()
+
+
+class SgmlRequestExtractor(BaseSgmlRequestExtractor):
+ """SGML Request Extractor"""
+
+ def __init__(self, tags=None, attrs=None):
+ """Initialize with custom tag & attribute function checkers"""
+ # defaults
+ tags = tuple(tags) if tags else ('a', 'area')
+ attrs = tuple(attrs) if attrs else ('href', )
+
+ tag_func = lambda x: x in tags
+ attr_func = lambda x: x in attrs
+ BaseSgmlRequestExtractor.__init__(self, tag=tag_func, attr=attr_func)
+
+# TODO: move to own file
+class XPathRequestExtractor(SgmlRequestExtractor):
+ """SGML Request Extractor with XPath restriction"""
+
+ def __init__(self, restrict_xpaths, tags=None, attrs=None):
+ """Initialize XPath restrictions"""
+ self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
+ SgmlRequestExtractor.__init__(self, tags, attrs)
+
+ def extract_requests(self, response):
+ """Restrict to XPath regions"""
+ hxs = HtmlXPathSelector(response)
+ fragments = (''.join(
+ html_frag for html_frag in hxs.select(xpath).extract()
+ ) for xpath in self.restrict_xpaths)
+ html_slice = ''.join(html_frag for html_frag in fragments)
+ return self._extract_requests(html_slice, response.url,
+ response.encoding)
+
diff --git a/scrapy/contrib_exp/crawlspider/reqgen.py b/scrapy/contrib_exp/crawlspider/reqgen.py
new file mode 100644
index 000000000..3858fbcf7
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/reqgen.py
@@ -0,0 +1,27 @@
+"""Request Generator"""
+from itertools import imap
+
+class RequestGenerator(object):
+ """Extracto and process requests from response"""
+
+ def __init__(self, req_extractors, req_processors, callback, spider=None):
+ """Initialize attributes"""
+ self._request_extractors = req_extractors
+ self._request_processors = req_processors
+ #TODO: resolve callback?
+ self._callback = callback
+
+ def generate_requests(self, response):
+ """Extract and process new requests from response.
+ Attach callback to each request as default callback."""
+ requests = []
+ for ext in self._request_extractors:
+ requests.extend(ext.extract_requests(response))
+
+ for proc in self._request_processors:
+ requests = proc(requests)
+
+ # return iterator
+ # @@@ creates new Request object with callback
+ return imap(lambda r: r.replace(callback=self._callback), requests)
+
diff --git a/scrapy/contrib_exp/crawlspider/reqproc.py b/scrapy/contrib_exp/crawlspider/reqproc.py
new file mode 100644
index 000000000..d39399b20
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/reqproc.py
@@ -0,0 +1,111 @@
+"""Request Processors"""
+from scrapy.utils.misc import arg_to_iter
+from scrapy.utils.url import canonicalize_url, url_is_from_any_domain
+
+from itertools import ifilter, imap
+
+import re
+
+class Canonicalize(object):
+ """Canonicalize Request Processor"""
+ def _replace_url(self, req):
+ # replace in-place
+ req.url = canonicalize_url(req.url)
+ return req
+
+ def __call__(self, requests):
+ """Canonicalize all requests' urls"""
+ return imap(self._replace_url, requests)
+
+
+class FilterDupes(object):
+ """Filter duplicate Requests"""
+
+ def __init__(self, *attributes):
+ """Initialize comparison attributes"""
+ self._attributes = tuple(attributes) if attributes \
+ else tuple(['url'])
+
+ def _equal_attr(self, obj1, obj2, attr):
+ return getattr(obj1, attr) == getattr(obj2, attr)
+
+ def _requests_equal(self, req1, req2):
+ """Attribute comparison helper"""
+ # look for not equal attribute
+ _not_equal = lambda attr: not self._equal_attr(req1, req2, attr)
+ for attr in ifilter(_not_equal, self._attributes):
+ return False
+ # all attributes equal
+ return True
+
+ def _request_in(self, request, requests_seen):
+ """Check if request is in given requests seen list"""
+ _req_seen = lambda r: self._requests_equal(r, request)
+ for seen in ifilter(_req_seen, requests_seen):
+ return True
+ # request not seen
+ return False
+
+ def __call__(self, requests):
+ """Filter seen requests"""
+ # per-call duplicates filter
+ self.requests_seen = set()
+ _not_seen = lambda r: not self._request_in(r, self.requests_seen)
+ for req in ifilter(_not_seen, requests):
+ yield req
+ # registry seen request
+ self.requests_seen.add(req)
+
+
+class FilterDomain(object):
+ """Filter request's domain"""
+
+ def __init__(self, allow=(), deny=()):
+ """Initialize allow/deny attributes"""
+ self.allow = tuple(arg_to_iter(allow))
+ self.deny = tuple(arg_to_iter(deny))
+
+ def __call__(self, requests):
+ """Filter domains"""
+ processed = (req for req in requests)
+
+ if self.allow:
+ processed = (req for req in requests
+ if url_is_from_any_domain(req.url, self.allow))
+ if self.deny:
+ processed = (req for req in requests
+ if not url_is_from_any_domain(req.url, self.deny))
+
+ return processed
+
+
+class FilterUrl(object):
+ """Filter request's url"""
+
+ def __init__(self, allow=(), deny=()):
+ """Initialize allow/deny attributes"""
+ _re_type = type(re.compile('', 0))
+
+ self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
+ for x in arg_to_iter(allow)]
+ self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
+ for x in arg_to_iter(deny)]
+
+ def __call__(self, requests):
+ """Filter request's url based on allow/deny rules"""
+ #TODO: filter valid urls here?
+ processed = (req for req in requests)
+
+ if self.allow_res:
+ processed = (req for req in requests
+ if self._matches(req.url, self.allow_res))
+ if self.deny_res:
+ processed = (req for req in requests
+ if not self._matches(req.url, self.deny_res))
+
+ return processed
+
+ def _matches(self, url, regexs):
+ """Returns True if url matches any regex in given list"""
+ return any(r.search(url) for r in regexs)
+
diff --git a/scrapy/contrib_exp/crawlspider/rules.py b/scrapy/contrib_exp/crawlspider/rules.py
new file mode 100644
index 000000000..ff1691ad0
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/rules.py
@@ -0,0 +1,100 @@
+"""Crawler Rules"""
+from scrapy.http import Request
+from scrapy.http import Response
+
+from functools import partial
+from itertools import ifilter
+
+from .matchers import BaseMatcher
+# default strint-to-matcher class
+from .matchers import UrlRegexMatcher
+
+class CompiledRule(object):
+ """Compiled version of Rule"""
+ def __init__(self, matcher, callback=None, follow=False):
+ """Initialize attributes checking type"""
+ assert isinstance(matcher, BaseMatcher)
+ assert callback is None or callable(callback)
+ assert isinstance(follow, bool)
+
+ self.matcher = matcher
+ self.callback = callback
+ self.follow = follow
+
+
+class Rule(object):
+ """Crawler Rule"""
+ def __init__(self, matcher=None, callback=None, follow=False, **kwargs):
+ """Store attributes"""
+ self.matcher = matcher
+ self.callback = callback
+ self.cb_kwargs = kwargs if kwargs else {}
+ self.follow = True if follow else False
+
+ if self.callback is None and self.follow is False:
+ raise ValueError("Rule must either have a callback or "
+ "follow=True: %r" % self)
+
+ def __repr__(self):
+ return "Rule(matcher=%r, callback=%r, follow=%r, **%r)" \
+ % (self.matcher, self.callback, self.follow, self.cb_kwargs)
+
+
+class RulesManager(object):
+ """Rules Manager"""
+ def __init__(self, rules, spider, default_matcher=UrlRegexMatcher):
+ """Initialize rules using spider and default matcher"""
+ self._rules = tuple()
+
+ # compile absolute/relative-to-spider callbacks"""
+ for rule in rules:
+ # prepare matcher
+ if rule.matcher is None:
+ # instance BaseMatcher by default
+ matcher = BaseMatcher()
+ elif isinstance(rule.matcher, BaseMatcher):
+ matcher = rule.matcher
+ else:
+ # matcher not BaseMatcher, check for string
+ if isinstance(rule.matcher, basestring):
+ # instance default matcher
+ matcher = default_matcher(rule.matcher)
+ else:
+ raise ValueError('Not valid matcher given %r in %r' \
+ % (rule.matcher, rule))
+
+ # prepare callback
+ if callable(rule.callback):
+ callback = rule.callback
+ elif not rule.callback is None:
+ # callback from spider
+ callback = getattr(spider, rule.callback)
+
+ if not callable(callback):
+ raise AttributeError('Invalid callback %r can not be resolved' \
+ % callback)
+ else:
+ callback = None
+
+ if rule.cb_kwargs:
+ # build partial callback
+ callback = partial(callback, **rule.cb_kwargs)
+
+ # append compiled rule to rules list
+ crule = CompiledRule(matcher, callback, follow=rule.follow)
+ self._rules += (crule, )
+
+ def get_rule_from_request(self, request):
+ """Returns first rule that matches given Request"""
+ _matches = lambda r: r.matcher.matches_request(request)
+ for rule in ifilter(_matches, self._rules):
+ # return first match of iterator
+ return rule
+
+ def get_rule_from_response(self, response):
+ """Returns first rule that matches given Response"""
+ _matches = lambda r: r.matcher.matches_response(response)
+ for rule in ifilter(_matches, self._rules):
+ # return first match of iterator
+ return rule
+
diff --git a/scrapy/contrib_exp/crawlspider/spider.py b/scrapy/contrib_exp/crawlspider/spider.py
new file mode 100644
index 000000000..730ad0e8d
--- /dev/null
+++ b/scrapy/contrib_exp/crawlspider/spider.py
@@ -0,0 +1,69 @@
+"""CrawlSpider v2"""
+from scrapy.spider import BaseSpider
+from scrapy.utils.spider import iterate_spider_output
+
+from .matchers import UrlListMatcher
+from .rules import Rule, RulesManager
+from .reqext import SgmlRequestExtractor
+from .reqgen import RequestGenerator
+from .reqproc import Canonicalize, FilterDupes
+
+class CrawlSpider(BaseSpider):
+ """CrawlSpider v2"""
+
+ request_extractors = None
+ request_processors = None
+ rules = []
+
+ def __init__(self):
+ """Initialize dispatcher"""
+ super(CrawlSpider, self).__init__()
+
+ # auto follow start urls
+ if self.start_urls:
+ _matcher = UrlListMatcher(self.start_urls)
+ # append new rule using type from current self.rules
+ rules = self.rules + type(self.rules)([
+ Rule(_matcher, follow=True)
+ ])
+ else:
+ rules = self.rules
+
+ # set defaults if not set
+ if self.request_extractors is None:
+ # default link extractor. Extracts all links from response
+ self.request_extractors = [ SgmlRequestExtractor() ]
+
+ if self.request_processors is None:
+ # default proccessor. Filter duplicates requests
+ self.request_processors = [ FilterDupes() ]
+
+
+ # wrap rules
+ self._rulesman = RulesManager(rules, spider=self)
+ # generates new requests with given callback
+ self._reqgen = RequestGenerator(self.request_extractors,
+ self.request_processors,
+ callback=self.parse)
+
+ def parse(self, response):
+ """Dispatch callback and generate requests"""
+ # get rule for response
+ rule = self._rulesman.get_rule_from_response(response)
+
+ if rule:
+ # dispatch callback if set
+ if rule.callback:
+ output = iterate_spider_output(rule.callback(response))
+ for req_or_item in output:
+ yield req_or_item
+
+ if rule.follow:
+ for req in self._reqgen.generate_requests(response):
+ # only dispatch request if has matching rule
+ if self._rulesman.get_rule_from_request(req):
+ yield req
+ else:
+ self.log("No rule for response %s" % response, level=log.WARNING)
+
+
diff --git a/scrapy/contrib_exp/pipeline/shoveitem.py b/scrapy/contrib_exp/pipeline/shoveitem.py
deleted file mode 100644
index 869c62c58..000000000
--- a/scrapy/contrib_exp/pipeline/shoveitem.py
+++ /dev/null
@@ -1,55 +0,0 @@
-"""
-A pipeline to persist objects using shove.
-
-Shove is a "new generation" shelve. For more information see:
-http://pypi.python.org/pypi/shove
-"""
-
-from string import Template
-
-from shove import Shove
-from scrapy.xlib.pydispatch import dispatcher
-
-from scrapy import log
-from scrapy.core import signals
-from scrapy.conf import settings
-from scrapy.core.exceptions import NotConfigured
-
-class ShoveItemPipeline(object):
-
- def __init__(self):
- self.uritpl = settings['SHOVEITEM_STORE_URI']
- if not self.uritpl:
- raise NotConfigured
- self.opts = settings['SHOVEITEM_STORE_OPT'] or {}
- self.stores = {}
-
- dispatcher.connect(self.spider_opened, signal=signals.spider_opened)
- dispatcher.connect(self.spider_closed, signal=signals.spider_closed)
-
- def process_item(self, spider, item):
- guid = str(item.guid)
-
- if guid in self.stores[spider]:
- if self.stores[spider][guid] == item:
- status = 'old'
- else:
- status = 'upd'
- else:
- status = 'new'
-
- if not status == 'old':
- self.stores[spider][guid] = item
- self.log(spider, item, status)
- return item
-
- def spider_opened(self, spider):
- uri = Template(self.uritpl).substitute(domain=spider.domain_name)
- self.stores[spider] = Shove(uri, **self.opts)
-
- def spider_closed(self, spider):
- self.stores[spider].sync()
-
- def log(self, spider, item, status):
- log.msg("Shove (%s): Item guid=%s" % (status, item.guid), level=log.DEBUG, \
- spider=spider)
diff --git a/scrapy/core/downloader/manager.py b/scrapy/core/downloader/manager.py
index b53db5811..aec17b05e 100644
--- a/scrapy/core/downloader/manager.py
+++ b/scrapy/core/downloader/manager.py
@@ -2,6 +2,7 @@
Download web pages using asynchronous IO
"""
+import random
from time import time
from twisted.internet import reactor, defer
@@ -20,15 +21,21 @@ class SpiderInfo(object):
def __init__(self, download_delay=None, max_concurrent_requests=None):
if download_delay is None:
- self.download_delay = settings.getfloat('DOWNLOAD_DELAY')
+ self._download_delay = settings.getfloat('DOWNLOAD_DELAY')
else:
- self.download_delay = download_delay
- if self.download_delay:
+ self._download_delay = float(download_delay)
+ if self._download_delay:
self.max_concurrent_requests = 1
elif max_concurrent_requests is None:
self.max_concurrent_requests = settings.getint('CONCURRENT_REQUESTS_PER_SPIDER')
else:
self.max_concurrent_requests = max_concurrent_requests
+ if self._download_delay and settings.getbool('RANDOMIZE_DOWNLOAD_DELAY'):
+ # same policy as wget --random-wait
+ self.random_delay_interval = (0.5*self._download_delay, \
+ 1.5*self._download_delay)
+ else:
+ self.random_delay_interval = None
self.active = set()
self.queue = []
@@ -44,6 +51,12 @@ class SpiderInfo(object):
# use self.active to include requests in the downloader middleware
return len(self.active) > 2 * self.max_concurrent_requests
+ def download_delay(self):
+ if self.random_delay_interval:
+ return random.uniform(*self.random_delay_interval)
+ else:
+ return self._download_delay
+
def cancel_request_calls(self):
for call in self.next_request_calls:
call.cancel()
@@ -99,8 +112,9 @@ class Downloader(object):
# Delay queue processing if a download_delay is configured
now = time()
- if site.download_delay:
- penalty = site.download_delay - now + site.lastseen
+ delay = site.download_delay()
+ if delay:
+ penalty = delay - now + site.lastseen
if penalty > 0:
d = defer.Deferred()
d.addCallback(self._process_queue)
diff --git a/scrapy/crawler.py b/scrapy/crawler.py
deleted file mode 100644
index e793d9125..000000000
--- a/scrapy/crawler.py
+++ /dev/null
@@ -1,66 +0,0 @@
-"""
-Crawler class
-
-The Crawler class can be used to crawl pages using the Scrapy crawler from
-outside a Scrapy project, for example, from a standalone script.
-
-To use it, instantiate it and call the "crawl" method with one (or more)
-requests. For example:
-
- >>> from scrapy.crawler import Crawler
- >>> from scrapy.http import Request
- >>> def parse_response(response):
- ... print "Visited: %s" % response.url
- ...
- >>> request = Request('http://scrapy.org', callback=parse_response)
- >>> crawler = Crawler()
- >>> crawler.crawl(request)
- Visited: http://scrapy.org
- >>>
-
-Request callbacks follow the same API of spiders callback, which means that all
-requests returned from the callbacks will be followed.
-
-See examples/scripts/count_and_follow_links.py for a more detailed example.
-
-WARNING: The Crawler class currently has a big limitation - it cannot be used
-more than once in the same Python process. This is due to the fact that Twisted
-reactors cannot be restarted. Hopefully, this limitation will be removed in the
-future.
-"""
-
-from scrapy.xlib.pydispatch import dispatcher
-from scrapy.core.manager import scrapymanager
-from scrapy.core.engine import scrapyengine
-from scrapy.conf import settings as scrapy_settings
-from scrapy import log
-
-class Crawler(object):
-
- def __init__(self, enable_log=False, stop_on_error=False, silence_errors=False, \
- settings=None):
- self.stop_on_error = stop_on_error
- self.silence_errors = silence_errors
- # disable offsite middleware (by default) because it prevents free crawling
- if settings is not None:
- settings.overrides.update(settings)
- scrapy_settings.overrides['SPIDER_MIDDLEWARES'] = {
- 'scrapy.contrib.spidermiddleware.offsite.OffsiteMiddleware': None}
- scrapy_settings.overrides['LOG_ENABLED'] = enable_log
- scrapymanager.configure()
- dispatcher.connect(self._logmessage_received, signal=log.logmessage_received)
-
- def crawl(self, *args):
- scrapymanager.runonce(*args)
-
- def stop(self):
- scrapyengine.stop()
- log.log_level = log.SILENT
- scrapyengine.kill()
-
- def _logmessage_received(self, message, level):
- if level <= log.ERROR:
- if not self.silence_errors:
- print "Crawler error: %s" % message
- if self.stop_on_error:
- self.stop()
diff --git a/scrapy/http/response/html.py b/scrapy/http/response/html.py
index f1557e6f7..dc812ac0e 100644
--- a/scrapy/http/response/html.py
+++ b/scrapy/http/response/html.py
@@ -23,9 +23,6 @@ class HtmlResponse(TextResponse):
METATAG_RE = re.compile(r'[\w-]+)')
XMLDECL_RE = re.compile(r'<\?xml\s.*?%s' % _encoding_re, re.I)
- def body_encoding(self):
- return self._body_declared_encoding() or super(XmlResponse, self).body_encoding()
-
@memoizemethod_noargs
def _body_declared_encoding(self):
chunk = self.body[:5000]
diff --git a/scrapy/log.py b/scrapy/log.py
index 71897ffee..c4f9287e5 100644
--- a/scrapy/log.py
+++ b/scrapy/log.py
@@ -29,8 +29,9 @@ BOT_NAME = settings['BOT_NAME']
# args: message, level, spider
logmessage_received = object()
-# default logging level
+# default values
log_level = DEBUG
+log_encoding = 'utf-8'
started = False
@@ -47,11 +48,12 @@ def _get_log_level(level_name_or_id=None):
def start(logfile=None, loglevel=None, logstdout=None):
"""Initialize and start logging facility"""
- global log_level, started
+ global log_level, log_encoding, started
if started or not settings.getbool('LOG_ENABLED'):
return
log_level = _get_log_level(loglevel)
+ log_encoding = settings['LOG_ENCODING']
started = True
# set log observer
@@ -74,7 +76,7 @@ def msg(message, level=INFO, component=BOT_NAME, domain=None, spider=None):
dispatcher.send(signal=logmessage_received, message=message, level=level, \
spider=spider)
system = domain or (spider.domain_name if spider else component)
- msg_txt = unicode_to_str("%s: %s" % (level_names[level], message))
+ msg_txt = unicode_to_str("%s: %s" % (level_names[level], message), log_encoding)
log.msg(msg_txt, system=system)
def exc(message, level=ERROR, component=BOT_NAME, domain=None, spider=None):
@@ -93,5 +95,5 @@ def err(_stuff=None, _why=None, **kwargs):
"use 'spider' argument instead", DeprecationWarning, stacklevel=2)
kwargs['system'] = domain or (spider.domain_name if spider else component)
if _why:
- _why = unicode_to_str("ERROR: %s" % _why)
+ _why = unicode_to_str("ERROR: %s" % _why, log_encoding)
log.err(_stuff, _why, **kwargs)
diff --git a/scrapy/mail.py b/scrapy/mail.py
index 8275e75cb..87825ed9c 100644
--- a/scrapy/mail.py
+++ b/scrapy/mail.py
@@ -47,34 +47,26 @@ class MailSender(object):
part = MIMEBase(*mimetype.split('/'))
part.set_payload(f.read())
Encoders.encode_base64(part)
- part.add_header('Content-Disposition', 'attachment; filename="%s"' % attach_name)
+ part.add_header('Content-Disposition', 'attachment; filename="%s"' \
+ % attach_name)
msg.attach(part)
else:
msg.set_payload(body)
- # FIXME ---------------------------------------------------------------------
- # There seems to be a problem with sending emails using deferreds when
- # the last thing left to do is sending the mail, cause the engine stops
- # the reactor and the email don't get send. we need to fix this. until
- # then, we'll revert to use Python standard (IO-blocking) smtplib.
-
- #dfd = self._sendmail(self.smtphost, self.mailfrom, rcpts, msg.as_string())
- #dfd.addCallbacks(self._sent_ok, self._sent_failed,
- # callbackArgs=[to, cc, subject, len(attachs)],
- # errbackArgs=[to, cc, subject, len(attachs)])
- import smtplib
- smtp = smtplib.SMTP(self.smtphost)
- smtp.sendmail(self.mailfrom, rcpts, msg.as_string())
- log.msg('Mail sent: To=%s Cc=%s Subject="%s"' % (to, cc, subject))
- smtp.close()
- # ---------------------------------------------------------------------------
+ dfd = self._sendmail(self.smtphost, self.mailfrom, rcpts, msg.as_string())
+ dfd.addCallbacks(self._sent_ok, self._sent_failed,
+ callbackArgs=[to, cc, subject, len(attachs)],
+ errbackArgs=[to, cc, subject, len(attachs)])
+ reactor.addSystemEventTrigger('before', 'shutdown', lambda: dfd)
def _sent_ok(self, result, to, cc, subject, nattachs):
- log.msg('Mail sent OK: To=%s Cc=%s Subject="%s" Attachs=%d' % (to, cc, subject, nattachs))
+ log.msg('Mail sent OK: To=%s Cc=%s Subject="%s" Attachs=%d' % \
+ (to, cc, subject, nattachs))
def _sent_failed(self, failure, to, cc, subject, nattachs):
errstr = str(failure.value)
- log.msg('Unable to send mail: To=%s Cc=%s Subject="%s" Attachs=%d - %s' % (to, cc, subject, nattachs, errstr), level=log.ERROR)
+ log.msg('Unable to send mail: To=%s Cc=%s Subject="%s" Attachs=%d - %s' % \
+ (to, cc, subject, nattachs, errstr), level=log.ERROR)
def _sendmail(self, smtphost, from_addr, to_addrs, msg, port=25):
""" This is based on twisted.mail.smtp.sendmail except that it
diff --git a/scrapy/selector/__init__.py b/scrapy/selector/__init__.py
index 37f86d431..bf2dc52c7 100644
--- a/scrapy/selector/__init__.py
+++ b/scrapy/selector/__init__.py
@@ -29,8 +29,8 @@ class XPathSelector(object_ref):
self.doc = Libxml2Document(response, factory=self._get_libxml2_doc)
self.xmlNode = self.doc.xmlDoc
elif text:
- response = TextResponse(url='about:blank', body=unicode_to_str(text), \
- encoding='utf-8')
+ response = TextResponse(url='about:blank', \
+ body=unicode_to_str(text, 'utf-8'), encoding='utf-8')
self.doc = Libxml2Document(response, factory=self._get_libxml2_doc)
self.xmlNode = self.doc.xmlDoc
self.expr = expr
diff --git a/scrapy/templates/project/module/pipelines.py.tmpl b/scrapy/templates/project/module/pipelines.py.tmpl
index fa6f5ea6f..e3f89342d 100644
--- a/scrapy/templates/project/module/pipelines.py.tmpl
+++ b/scrapy/templates/project/module/pipelines.py.tmpl
@@ -4,5 +4,5 @@
# See: http://doc.scrapy.org/topics/item-pipeline.html
class ${ProjectName}Pipeline(object):
- def process_item(self, domain, item):
+ def process_item(self, spider, item):
return item
diff --git a/scrapy/templates/spiders/crawl.tmpl b/scrapy/templates/spiders/crawl.tmpl
index 2d4f7fce6..18fd0ec42 100644
--- a/scrapy/templates/spiders/crawl.tmpl
+++ b/scrapy/templates/spiders/crawl.tmpl
@@ -10,15 +10,15 @@ class $classname(CrawlSpider):
start_urls = ['http://www.$site/']
rules = (
- Rule(SgmlLinkExtractor(allow=(r'Items/', )), 'parse_item', follow=True),
+ Rule(SgmlLinkExtractor(allow=r'Items/'), callback='parse_item', follow=True),
)
def parse_item(self, response):
- xs = HtmlXPathSelector(response)
+ hxs = HtmlXPathSelector(response)
i = ${ProjectName}Item()
- #i['site_id'] = xs.select('//input[@id="sid"]/@value').extract()
- #i['name'] = xs.select('//div[@id="name"]').extract()
- #i['description'] = xs.select('//div[@id="description"]').extract()
+ #i['site_id'] = hxs.select('//input[@id="sid"]/@value').extract()
+ #i['name'] = hxs.select('//div[@id="name"]').extract()
+ #i['description'] = hxs.select('//div[@id="description"]').extract()
return i
SPIDER = $classname()
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_matchers.py b/scrapy/tests/test_contrib_exp_crawlspider_matchers.py
new file mode 100644
index 000000000..4cb832aa1
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_matchers.py
@@ -0,0 +1,94 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import Response
+
+from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher
+
+import re
+
+class MatchersTest(unittest.TestCase):
+
+ def setUp(self):
+ pass
+
+ def test_base_matcher(self):
+ matcher = BaseMatcher()
+
+ request = Request('http://example.com')
+ response = Response('http://example.com')
+
+ self.assertTrue(matcher.matches_request(request))
+ self.assertTrue(matcher.matches_response(response))
+
+ def test_url_matcher(self):
+ matcher = UrlMatcher('http://example.com')
+
+ request = Request('http://example.com')
+ response = Response('http://example.com')
+
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_request(response))
+
+ request = Request('http://example2.com')
+ response = Response('http://example2.com')
+
+ self.failIf(matcher.matches_request(request))
+ self.failIf(matcher.matches_request(response))
+
+ def test_url_regex_matcher(self):
+ matcher = UrlRegexMatcher(r'sample')
+ urls = (
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample4.html',
+ )
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_response(response))
+
+ matcher = UrlRegexMatcher(r'sample_fail')
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failIf(matcher.matches_request(request))
+ self.failIf(matcher.matches_response(response))
+
+ matcher = UrlRegexMatcher(r'SAMPLE\d+', re.IGNORECASE)
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_response(response))
+
+ def test_url_list_matcher(self):
+ urls = (
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample4.html',
+ )
+ urls2 = (
+ 'http://example.com/sample5.html',
+ 'http://example.com/sample6.html',
+ 'http://example.com/sample7.html',
+ 'http://example.com/sample8.html',
+ 'http://example.com/',
+ )
+ matcher = UrlListMatcher(urls)
+
+ # match urls
+ for url in urls:
+ request, response = Request(url), Response(url)
+ self.failUnless(matcher.matches_request(request))
+ self.failUnless(matcher.matches_response(response))
+
+ # non-match urls
+ for url in urls2:
+ request, response = Request(url), Response(url)
+ self.failIf(matcher.matches_request(request))
+ self.failIf(matcher.matches_response(response))
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqext.py b/scrapy/tests/test_contrib_exp_crawlspider_reqext.py
new file mode 100644
index 000000000..0259b30fb
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_reqext.py
@@ -0,0 +1,156 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import HtmlResponse
+from scrapy.tests import get_testdata
+
+from scrapy.contrib_exp.crawlspider.reqext import BaseSgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqext import XPathRequestExtractor
+
+class AbstractRequestExtractorTest(unittest.TestCase):
+
+ def _requests_equals(self, list1, list2):
+ """Compares request's urls and link_text"""
+ for (r1, r2) in zip(list1, list2):
+ if r1.url != r2.url:
+ return False
+ if r1.meta['link_text'] != r2.meta['link_text']:
+ return False
+ # all equal
+ return True
+
+
+class RequestExtractorTest(AbstractRequestExtractorTest):
+
+ def test_basic(self):
+ base_url = 'http://example.org/somepage/index.html'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+ requests = [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ Request('http://example.org/about.html',
+ meta={'link_text': 'About us'}),
+ Request('http://example.org/othercat.html',
+ meta={'link_text': 'Other category'}),
+ Request('http://example.org/',
+ meta={'link_text': ''}),
+ ]
+
+ response = HtmlResponse(base_url, body=html)
+ reqx = BaseSgmlRequestExtractor() # default: tag=a, attr=href
+
+ self.failUnless(
+ self._requests_equals(requests, reqx.extract_requests(response))
+ )
+
+ def test_base_url(self):
+ reqx = BaseSgmlRequestExtractor()
+
+ html = """Page title
+
+ Item 12
+ """
+ response = HtmlResponse("https://example.org/p/index.html", body=html)
+ reqs = reqx.extract_requests(response)
+ self.failUnless(self._requests_equals( \
+ [Request('http://otherdomain.com/base/item/12.html', \
+ meta={'link_text': 'Item 12'})], reqs), reqs)
+
+ # base url is an absolute path and relative to host
+ html = """Page title
+
+ Item 12
+ """
+ response = HtmlResponse("https://example.org/p/index.html", body=html)
+ reqs = reqx.extract_requests(response)
+ self.failUnless(self._requests_equals( \
+ [Request('https://example.org/item/12.html', \
+ meta={'link_text': 'Item 12'})], reqs), reqs)
+
+ # base url has no scheme
+ html = """Page title
+
+ Item 12
+ """
+ response = HtmlResponse("https://example.org/p/index.html", body=html)
+ reqs = reqx.extract_requests(response)
+ self.failUnless(self._requests_equals( \
+ [Request('https://noscheme.com/base/item/12.html', \
+ meta={'link_text': 'Item 12'})], reqs), reqs)
+
+ def test_extraction_encoding(self):
+ #TODO: use own fixtures
+ body = get_testdata('link_extractor', 'linkextractor_noenc.html')
+ response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body,
+ headers={'Content-Type': ['text/html; charset=utf-8']})
+ response_noenc = HtmlResponse(url='http://example.com/noenc',
+ body=body)
+ body = get_testdata('link_extractor', 'linkextractor_latin1.html')
+ response_latin1 = HtmlResponse(url='http://example.com/latin1',
+ body=body)
+
+ reqx = BaseSgmlRequestExtractor()
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(response_utf8),
+ [ Request(url='http://example.com/sample_%C3%B1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample_%E2%82%AC.html',
+ meta={'link_text':
+ 'sample \xe2\x82\xac text'.decode('utf-8')}) ]
+ )
+ )
+
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(response_noenc),
+ [ Request(url='http://example.com/sample_%C3%B1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample_%E2%82%AC.html',
+ meta={'link_text':
+ 'sample \xe2\x82\xac text'.decode('utf-8')}) ]
+ )
+ )
+
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(response_latin1),
+ [ Request(url='http://example.com/sample_%F1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample_%E1.html',
+ meta={'link_text':
+ 'sample \xe1 text'.decode('latin1')}) ]
+ )
+ )
+
+
+class SgmlRequestExtractorTest(AbstractRequestExtractorTest):
+ pass
+
+
+class XPathRequestExtractorTest(AbstractRequestExtractorTest):
+
+ def setUp(self):
+ # TODO: use own fixtures
+ body = get_testdata('link_extractor', 'sgml_linkextractor.html')
+ self.response = HtmlResponse(url='http://example.com/index', body=body)
+
+
+ def test_restrict_xpaths(self):
+ reqx = XPathRequestExtractor('//div[@id="subwrapper"]')
+ self.failUnless(
+ self._requests_equals(
+ reqx.extract_requests(self.response),
+ [ Request(url='http://example.com/sample1.html',
+ meta={'link_text': ''}),
+ Request(url='http://example.com/sample2.html',
+ meta={'link_text': 'sample 2'}) ]
+ )
+ )
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py b/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py
new file mode 100644
index 000000000..67aca2387
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_reqgen.py
@@ -0,0 +1,128 @@
+from twisted.internet import defer
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import HtmlResponse
+from scrapy.utils.python import equal_attributes
+
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+from scrapy.contrib_exp.crawlspider.reqgen import RequestGenerator
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain
+from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
+
+class RequestGeneratorTest(unittest.TestCase):
+
+ def setUp(self):
+ url = 'http://example.org/somepage/index.html'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+
+ self.response = HtmlResponse(url, body=html)
+ self.deferred = defer.Deferred()
+ self.requests = [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ Request('http://example.org/about.html',
+ meta={'link_text': 'About us'}),
+ Request('http://example.org/othercat.html',
+ meta={'link_text': 'Other category'}),
+ Request('http://example.org/',
+ meta={'link_text': ''}),
+ ]
+
+ def _equal_requests_list(self, list1, list2):
+ list1 = list(list1)
+ list2 = list(list2)
+ if not len(list1) == len(list2):
+ return False
+
+ for (req1, req2) in zip(list1, list2):
+ if not equal_attributes(req1, req2, ['url']):
+ return False
+ return True
+
+ def test_basic(self):
+ reqgen = RequestGenerator([], [], callback=self.deferred)
+ # returns generator
+ requests = reqgen.generate_requests(self.response)
+ self.failUnlessEqual(list(requests), [])
+
+ def test_request_extractor(self):
+ extractors = [
+ SgmlRequestExtractor()
+ ]
+
+ # extract all requests
+ reqgen = RequestGenerator(extractors, [], callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+ self.failUnless(self._equal_requests_list(requests, self.requests))
+
+ for req in requests:
+ # check callback
+ self.failUnlessEqual(req.deferred, self.deferred)
+
+ def test_request_processor(self):
+ extractors = [
+ SgmlRequestExtractor()
+ ]
+
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+ self.failUnless(self._equal_requests_list(requests, self.requests))
+
+ # filter domain
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterDomain(deny='example.org'),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+ self.failUnlessEqual(list(requests), [])
+
+ # filter url
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterUrl(deny=(r'about', r'othercat')),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+
+ self.failUnless(self._equal_requests_list(requests, [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ Request('http://example.org/',
+ meta={'link_text': ''}),
+ ]))
+
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ FilterUrl(allow=r'/somepage/'),
+ ]
+
+ reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
+ requests = reqgen.generate_requests(self.response)
+
+ self.failUnless(self._equal_requests_list(requests, [
+ Request('http://example.org/somepage/item/12.html',
+ meta={'link_text': 'Item 12'}),
+ ]))
+
+
+
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py b/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py
new file mode 100644
index 000000000..da5db67b2
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_reqproc.py
@@ -0,0 +1,144 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain
+from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
+
+import copy
+
+class RequestProcessorsTest(unittest.TestCase):
+
+ def test_canonicalize_requests(self):
+ urls = [
+ 'http://example.com/do?&b=1&a=2&c=3',
+ 'http://example.com/do?123,&q=a space',
+ ]
+ urls_after = [
+ 'http://example.com/do?a=2&b=1&c=3',
+ 'http://example.com/do?123%2C=&q=a+space',
+ ]
+
+ proc = Canonicalize()
+ results = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(results, urls_after)
+
+ def test_unique_requests(self):
+ urls = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ ]
+ urls_unique = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ ]
+
+ proc = FilterDupes()
+ results = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(results, urls_unique)
+
+ # Check custom attributes
+ requests = [
+ Request('http://example.com', method='GET'),
+ Request('http://example.com', method='POST'),
+ ]
+ proc = FilterDupes('url', 'method')
+ self.failUnlessEqual(len(list(proc(requests))), 2)
+
+ proc = FilterDupes('url')
+ self.failUnlessEqual(len(list(proc(requests))), 1)
+
+ def test_filter_domain(self):
+ urls = [
+ 'http://blah1.com/index',
+ 'http://blah2.com/index',
+ 'http://blah1.com/section',
+ 'http://blah2.com/section',
+ ]
+
+ proc = FilterDomain(allow=('blah1.com'), deny=('blah2.com'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [
+ 'http://blah1.com/index',
+ 'http://blah1.com/section',
+ ])
+
+ proc = FilterDomain(deny=('blah1.com', 'blah2.com'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [])
+
+ proc = FilterDomain(allow=('blah1.com', 'blah2.com'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, urls)
+
+ def test_filter_url(self):
+ urls = [
+ 'http://blah1.com/index',
+ 'http://blah2.com/index',
+ 'http://blah1.com/section',
+ 'http://blah2.com/section',
+ ]
+
+ proc = FilterUrl(allow=(r'blah1'), deny=(r'blah2'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [
+ 'http://blah1.com/index',
+ 'http://blah1.com/section',
+ ])
+
+ proc = FilterUrl(deny=('blah1', 'blah2'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, [])
+
+ proc = FilterUrl(allow=('index$', 'section$'))
+ filtered = [req.url for req in proc(Request(url) for url in urls)]
+ self.failUnlessEquals(filtered, urls)
+
+
+
+ def test_all_processors(self):
+ urls = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/do?&b=1&a=2&c=3',
+ 'http://example.com/do?123,&q=a space',
+ ]
+ urls_processed = [
+ 'http://example.com/sample1.html',
+ 'http://example.com/sample2.html',
+ 'http://example.com/sample3.html',
+ 'http://example.com/do?a=2&b=1&c=3',
+ 'http://example.com/do?123%2C=&q=a+space',
+ ]
+
+ processors = [
+ Canonicalize(),
+ FilterDupes(),
+ ]
+
+ def _process(requests):
+ """Apply all processors"""
+ # copy list
+ processed = [copy.copy(req) for req in requests]
+ for proc in processors:
+ processed = proc(processed)
+ return processed
+
+ # empty requests
+ results1 = [r.url for r in _process([])]
+ self.failUnlessEquals(results1, [])
+
+ # try urls
+ requests = (Request(url) for url in urls)
+ results2 = [r.url for r in _process(requests)]
+ self.failUnlessEquals(results2, urls_processed)
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_rules.py b/scrapy/tests/test_contrib_exp_crawlspider_rules.py
new file mode 100644
index 000000000..0fbe52415
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_rules.py
@@ -0,0 +1,262 @@
+from twisted.trial import unittest
+
+from scrapy.http import HtmlResponse
+from scrapy.spider import BaseSpider
+from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
+
+from scrapy.contrib_exp.crawlspider.rules import CompiledRule
+from scrapy.contrib_exp.crawlspider.rules import Rule
+from scrapy.contrib_exp.crawlspider.rules import RulesManager
+
+from functools import partial
+
+class RuleInitializationTest(unittest.TestCase):
+
+ def test_fail_if_rule_null(self):
+ # fail on empty rule
+ self.failUnlessRaises(ValueError, Rule)
+ self.failUnlessRaises(ValueError, Rule,
+ **dict(callback=None, follow=None))
+ self.failUnlessRaises(ValueError, Rule,
+ **dict(callback=None, follow=False))
+
+ def test_minimal_arguments_to_instantiation(self):
+ # not fail if callback set
+ self.failUnless(Rule(callback=lambda: True))
+ # not fail if follow set
+ self.failUnless(Rule(follow=True))
+
+ def test_validate_default_attributes(self):
+ # test null Rule
+ rule = Rule(follow=True)
+ self.failUnlessEqual(None, rule.matcher)
+ self.failUnlessEqual(None, rule.callback)
+ self.failUnlessEqual({}, rule.cb_kwargs)
+ # follow default False
+ self.failUnlessEqual(True, rule.follow)
+
+ def test_validate_attributes_set(self):
+ matcher = BaseMatcher()
+ callback = lambda: True
+ rule = Rule(matcher, callback, True, a=1)
+ # test attributes
+ self.failUnlessEqual(matcher, rule.matcher)
+ self.failUnlessEqual(callback, rule.callback)
+ self.failUnlessEqual({'a': 1}, rule.cb_kwargs)
+ self.failUnlessEqual(True, rule.follow)
+
+class CompiledRuleInitializationTest(unittest.TestCase):
+
+ def test_fail_on_invalid_matcher(self):
+ # pass with valid matcher
+ self.failUnless(CompiledRule(BaseMatcher()),
+ "Failed CompiledRule instantiation")
+
+ # at least needs valid matcher
+ self.assertRaises(AssertionError, CompiledRule, None)
+ self.assertRaises(AssertionError, CompiledRule, False)
+ self.assertRaises(AssertionError, CompiledRule, True)
+
+ def test_fail_on_invalid_callback(self):
+ # pass with valid callback
+ callback = lambda: True
+ self.failUnless(CompiledRule(BaseMatcher(), callback))
+ # pass with callback none
+ self.failUnless(CompiledRule(BaseMatcher(), None))
+
+ # assert on invalid callback
+ self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
+ 'myfunc')
+
+ # numeric variable
+ var = 123
+ self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
+ var)
+
+ class A:
+ pass
+
+ # random instance
+ self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
+ A())
+
+
+ def test_fail_on_invalid_follow_value(self):
+ callback = lambda: True
+ matcher = BaseMatcher()
+ # pass bool
+ self.failUnless(CompiledRule(matcher, callback, True))
+ self.failUnless(CompiledRule(matcher, callback, False))
+
+ # assert with non-bool
+ self.assertRaises(AssertionError, CompiledRule, matcher,
+ callback, None)
+ self.assertRaises(AssertionError, CompiledRule, matcher,
+ callback, 1)
+
+ def test_validate_default_attributes(self):
+ callback = lambda: True
+ matcher = BaseMatcher()
+ rule = CompiledRule(matcher, callback, True)
+
+ # test attributes
+ self.failUnlessEqual(matcher, rule.matcher)
+ self.failUnlessEqual(callback, rule.callback)
+ self.failUnlessEqual(True, rule.follow)
+
+
+class RulesTest(unittest.TestCase):
+ def test_rules_manager_basic(self):
+ spider = BaseSpider()
+ response1 = HtmlResponse('http://example.org')
+ response2 = HtmlResponse('http://othersite.org')
+ rulesman = RulesManager([], spider)
+
+ # should return none
+ self.failIf(rulesman.get_rule_from_response(response1))
+ self.failIf(rulesman.get_rule_from_response(response2))
+
+ # rules manager with match-all rule
+ rulesman = RulesManager([
+ Rule(BaseMatcher(), follow=True),
+ ], spider)
+
+ # returns CompiledRule
+ rule1 = rulesman.get_rule_from_response(response1)
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnless(isinstance(rule1, CompiledRule))
+ self.failUnless(isinstance(rule2, CompiledRule))
+ self.assert_(rule1 is rule2)
+ self.failUnlessEqual(rule1.callback, None)
+ self.failUnlessEqual(rule1.follow, True)
+
+ def test_rules_manager_empty_rule(self):
+ spider = BaseSpider()
+ response = HtmlResponse('http://example.org')
+
+ rulesman = RulesManager([Rule(follow=True)], spider)
+
+ rule = rulesman.get_rule_from_response(response)
+ # default matcher if None: BaseMatcher
+ self.failUnless(isinstance(rule.matcher, BaseMatcher))
+
+ def test_rules_manager_default_matcher(self):
+ spider = BaseSpider()
+ response = HtmlResponse('http://example.org')
+ callback = lambda x: None
+
+ rulesman = RulesManager([
+ Rule('http://example.org', callback),
+ ], spider, default_matcher=UrlMatcher)
+
+ rule = rulesman.get_rule_from_response(response)
+ self.failUnless(isinstance(rule.matcher, UrlMatcher))
+
+ def test_rules_manager_matchers(self):
+ spider = BaseSpider()
+ response1 = HtmlResponse('http://example.org')
+ response2 = HtmlResponse('http://othersite.org')
+
+ urlmatcher = UrlMatcher('http://example.org')
+ basematcher = BaseMatcher()
+ # callback needed for Rule
+ callback = lambda x: None
+
+ # test fail matcher resolve
+ self.assertRaises(ValueError, RulesManager,
+ [Rule(False, callback)], spider)
+ self.assertRaises(ValueError, RulesManager,
+ [Rule(spider, callback)], spider)
+
+ rulesman = RulesManager([
+ Rule(urlmatcher, callback),
+ Rule(basematcher, callback),
+ ], spider)
+
+ # response1 matches example.org
+ rule1 = rulesman.get_rule_from_response(response1)
+ # response2 is catch by BaseMatcher()
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnlessEqual(rule1.matcher, urlmatcher)
+ self.failUnlessEqual(rule2.matcher, basematcher)
+
+ # reverse order. BaseMatcher should match all
+ rulesman = RulesManager([
+ Rule(basematcher, callback),
+ Rule(urlmatcher, callback),
+ ], spider)
+
+ rule1 = rulesman.get_rule_from_response(response1)
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnlessEqual(rule1.matcher, basematcher)
+ self.failUnlessEqual(rule2.matcher, basematcher)
+ self.failUnless(rule1 is rule2)
+
+ def test_rules_manager_callbacks(self):
+ mycallback = lambda: True
+
+ spider = BaseSpider()
+ spider.parse_item = lambda: True
+
+ response1 = HtmlResponse('http://example.org')
+ response2 = HtmlResponse('http://othersite.org')
+
+ rulesman = RulesManager([
+ Rule('example', mycallback),
+ Rule('othersite', 'parse_item'),
+ ], spider, default_matcher=UrlRegexMatcher)
+
+ rule1 = rulesman.get_rule_from_response(response1)
+ rule2 = rulesman.get_rule_from_response(response2)
+
+ self.failUnlessEqual(rule1.callback, mycallback)
+ self.failUnlessEqual(rule2.callback, spider.parse_item)
+
+ # fail unknown callback
+ self.assertRaises(AttributeError, RulesManager, [
+ Rule(BaseMatcher(), 'mycallback')
+ ], spider)
+ # fail not callable
+ spider.not_callable = True
+ self.assertRaises(AttributeError, RulesManager, [
+ Rule(BaseMatcher(), 'not_callable')
+ ], spider)
+
+
+ def test_rules_manager_callback_with_arguments(self):
+ spider = BaseSpider()
+ response = HtmlResponse('http://example.org')
+
+ kwargs = {'a': 1}
+
+ def myfunc(**mykwargs):
+ return mykwargs
+
+ # verify return validation
+ self.failUnlessEquals(kwargs, myfunc(**kwargs))
+
+ # test callback w/o arguments
+ rulesman = RulesManager([
+ Rule(BaseMatcher(), myfunc),
+ ], spider)
+ rule = rulesman.get_rule_from_response(response)
+
+ # without arguments should return same callback
+ self.failUnlessEqual(rule.callback, myfunc)
+
+ # test callback w/ arguments
+ rulesman = RulesManager([
+ Rule(BaseMatcher(), myfunc, **kwargs),
+ ], spider)
+ rule = rulesman.get_rule_from_response(response)
+
+ # with argument should return partial applied callback
+ self.failUnless(isinstance(rule.callback, partial))
+ self.failUnlessEquals(kwargs, rule.callback())
+
+
diff --git a/scrapy/tests/test_contrib_exp_crawlspider_spider.py b/scrapy/tests/test_contrib_exp_crawlspider_spider.py
new file mode 100644
index 000000000..5e067508a
--- /dev/null
+++ b/scrapy/tests/test_contrib_exp_crawlspider_spider.py
@@ -0,0 +1,222 @@
+from twisted.trial import unittest
+
+from scrapy.http import Request
+from scrapy.http import HtmlResponse
+from scrapy.item import BaseItem
+from scrapy.utils.spider import iterate_spider_output
+
+# basics
+from scrapy.contrib_exp.crawlspider import CrawlSpider
+from scrapy.contrib_exp.crawlspider import Rule
+
+# matchers
+from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
+from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher
+
+# extractors
+from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
+
+# processors
+from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
+from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
+
+
+# mock items
+class Item1(BaseItem):
+ pass
+
+class Item2(BaseItem):
+ pass
+
+class Item3(BaseItem):
+ pass
+
+
+class CrawlSpiderTest(unittest.TestCase):
+
+ def spider_factory(self, rules=[],
+ extractors=[], processors=[],
+ start_urls=[]):
+ # mock spider
+ class Spider(CrawlSpider):
+ def parse_item1(self, response):
+ return Item1()
+
+ def parse_item2(self, response):
+ return Item2()
+
+ def parse_item3(self, response):
+ return Item3()
+
+ def parse_request1(self, response):
+ return Request('http://example.org/request1')
+
+ def parse_request2(self, response):
+ return Request('http://example.org/request2')
+
+ Spider.start_urls = start_urls
+ Spider.rules = rules
+ Spider.request_extractors = extractors
+ Spider.request_processors = processors
+
+ return Spider()
+
+ def test_start_url_auto_rule(self):
+ spider = self.spider_factory()
+ # zero spider rules
+ self.failUnlessEqual(len(spider.rules), 0)
+ self.failUnlessEqual(len(spider._rulesman._rules), 0)
+
+ spider = self.spider_factory(start_urls=['http://example.org'])
+
+ self.failUnlessEqual(len(spider.rules), 0)
+ self.failUnlessEqual(len(spider._rulesman._rules), 1)
+
+ def test_start_url_matcher(self):
+ url = 'http://example.org'
+ spider = self.spider_factory(start_urls=[url])
+
+ response = HtmlResponse(url)
+
+ rule = spider._rulesman.get_rule_from_response(response)
+ self.failUnless(isinstance(rule.matcher, UrlListMatcher))
+
+ response = HtmlResponse(url + '/item.html')
+
+ rule = spider._rulesman.get_rule_from_response(response)
+ self.failUnless(rule is None)
+
+ # TODO: remove this block
+ # in previous version get_rule returns rule from response.request
+ response.request = Request(url)
+ rule = spider._rulesman.get_rule_from_response(response.request)
+ self.failUnless(isinstance(rule.matcher, UrlListMatcher))
+ self.failUnlessEqual(rule.follow, True)
+
+ def test_parse_callback(self):
+ response = HtmlResponse('http://example.org')
+ rules = (
+ Rule(BaseMatcher(), 'parse_item1'),
+ )
+ spider = self.spider_factory(rules)
+
+ result = list(spider.parse(response))
+ self.failUnlessEqual(len(result), 1)
+ self.failUnless(isinstance(result[0], Item1))
+
+ def test_crawling_start_url(self):
+ url = 'http://example.org/'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+ response = HtmlResponse(url, body=html)
+
+ extractors = (SgmlRequestExtractor(), )
+ spider = self.spider_factory(start_urls=[url],
+ extractors=extractors)
+ result = list(spider.parse(response))
+
+ # 1 request extracted: example.org/
+ # because requests returns only matching
+ self.failUnlessEqual(len(result), 1)
+
+ # we will add catch-all rule to extract all
+ callback = lambda x: None
+ rules = [Rule(r'\.html$', callback=callback)]
+ spider = self.spider_factory(rules, start_urls=[url],
+ extractors=extractors)
+ result = list(spider.parse(response))
+
+ # 4 requests extracted
+ # 3 of .html pattern
+ # 1 of start url patter
+ self.failUnlessEqual(len(result), 4)
+
+ def test_crawling_simple_rule(self):
+ url = 'http://example.org/somepage/index.html'
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+
+ response = HtmlResponse(url, body=html)
+
+ rules = (
+ # first response callback
+ Rule(r'index\.html', 'parse_item1'),
+ )
+ spider = self.spider_factory(rules)
+ result = list(spider.parse(response))
+
+ # should return Item1
+ self.failUnlessEqual(len(result), 1)
+ self.failUnless(isinstance(result[0], Item1))
+
+ # test request generation
+ rules = (
+ # first response without callback and follow flag
+ Rule(r'index\.html', follow=True),
+ Rule(r'(\.html|/)$', 'parse_item1'),
+ )
+ spider = self.spider_factory(rules)
+ result = list(spider.parse(response))
+
+ # 0 because spider does not have extractors
+ self.failUnlessEqual(len(result), 0)
+
+ extractors = (SgmlRequestExtractor(), )
+
+ # instance spider with extractor
+ spider = self.spider_factory(rules, extractors)
+ result = list(spider.parse(response))
+ # 4 requests extracted
+ self.failUnlessEqual(len(result), 4)
+
+ def test_crawling_multiple_rules(self):
+ html = """Page title
+ Item 12
+ About us
+
+ Other category
+
"""
+
+ response = HtmlResponse('http://example.org/index.html', body=html)
+ response1 = HtmlResponse('http://example.org/1.html')
+ response2 = HtmlResponse('http://example.org/othercat.html')
+
+ rules = (
+ Rule(r'\d+\.html$', 'parse_item1'),
+ Rule(r'othercat\.html$', 'parse_item2'),
+ # follow-only rules
+ Rule(r'index\.html', 'parse_item3', follow=True)
+ )
+ extractors = [SgmlRequestExtractor()]
+ spider = self.spider_factory(rules, extractors)
+
+ result = list(spider.parse(response))
+ # 1 Item 2 Requests
+ self.failUnlessEqual(len(result), 3)
+ # parse_item3
+ self.failUnless(isinstance(result[0], Item3))
+ only_requests = lambda r: isinstance(r, Request)
+ requests = filter(only_requests, result[1:])
+ self.failUnlessEqual(len(requests), 2)
+ self.failUnless(all(requests))
+
+ result1 = list(spider.parse(response1))
+ # parse_item1
+ self.failUnlessEqual(len(result1), 1)
+ self.failUnless(isinstance(result1[0], Item1))
+
+ result2 = list(spider.parse(response2))
+ # parse_item2
+ self.failUnlessEqual(len(result2), 1)
+ self.failUnless(isinstance(result2[0], Item2))
+
+
diff --git a/scrapy/tests/test_contrib_linkextractors.py b/scrapy/tests/test_contrib_linkextractors.py
index 4a60b8a2d..65ffad714 100644
--- a/scrapy/tests/test_contrib_linkextractors.py
+++ b/scrapy/tests/test_contrib_linkextractors.py
@@ -35,6 +35,20 @@ class LinkExtractorTestCase(unittest.TestCase):
self.assertEqual(lx.extract_links(response),
[Link(url='http://otherdomain.com/base/item/12.html', text='Item 12')])
+ # base url is an absolute path and relative to host
+ html = """Page title
+ Item 12
"""
+ response = HtmlResponse("https://example.org/somepage/index.html", body=html)
+ self.assertEqual(lx.extract_links(response),
+ [Link(url='https://example.org/item/12.html', text='Item 12')])
+
+ # base url has no scheme
+ html = """Page title
+ Item 12
"""
+ response = HtmlResponse("https://example.org/somepage/index.html", body=html)
+ self.assertEqual(lx.extract_links(response),
+ [Link(url='https://noschemedomain.com/path/to/item/12.html', text='Item 12')])
+
def test_extraction_encoding(self):
body = get_testdata('link_extractor', 'linkextractor_noenc.html')
response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']})
diff --git a/scrapy/tests/test_downloadermiddleware_redirect.py b/scrapy/tests/test_downloadermiddleware_redirect.py
index ba7daece8..f9a93b910 100644
--- a/scrapy/tests/test_downloadermiddleware_redirect.py
+++ b/scrapy/tests/test_downloadermiddleware_redirect.py
@@ -3,7 +3,7 @@ import unittest
from scrapy.contrib.downloadermiddleware.redirect import RedirectMiddleware
from scrapy.spider import BaseSpider
from scrapy.core.exceptions import IgnoreRequest
-from scrapy.http import Request, Response, Headers
+from scrapy.http import Request, Response, HtmlResponse, Headers
class RedirectMiddlewareTest(unittest.TestCase):
@@ -58,7 +58,7 @@ class RedirectMiddlewareTest(unittest.TestCase):
"""
req = Request(url='http://example.org')
- rsp = Response(url='http://example.org', body=body)
+ rsp = HtmlResponse(url='http://example.org', body=body)
req2 = self.mw.process_response(req, rsp, self.spider)
assert isinstance(req2, Request)
@@ -70,7 +70,7 @@ class RedirectMiddlewareTest(unittest.TestCase):
"""
req = Request(url='http://example.org')
- rsp = Response(url='http://example.org', body=body)
+ rsp = HtmlResponse(url='http://example.org', body=body)
rsp2 = self.mw.process_response(req, rsp, self.spider)
assert rsp is rsp2
@@ -81,7 +81,7 @@ class RedirectMiddlewareTest(unittest.TestCase):
"""
req = Request(url='http://example.org', method='POST', body='test',
headers={'Content-Type': 'text/plain', 'Content-length': '4'})
- rsp = Response(url='http://example.org', body=body)
+ rsp = HtmlResponse(url='http://example.org', body=body)
req2 = self.mw.process_response(req, rsp, self.spider)
assert isinstance(req2, Request)
diff --git a/scrapy/tests/test_encoding_aliases.py b/scrapy/tests/test_encoding_aliases.py
deleted file mode 100644
index cc3a0089f..000000000
--- a/scrapy/tests/test_encoding_aliases.py
+++ /dev/null
@@ -1,21 +0,0 @@
-import unittest
-
-import scrapy # adds encoding aliases (if not added before)
-
-class EncodingAliasesTestCase(unittest.TestCase):
-
- def test_encoding_aliases(self):
- """Test common encdoing aliases not included in Python"""
-
- uni = u'\u041c\u041e\u0421K\u0412\u0410'
- str = uni.encode('windows-1251')
- self.assertEqual(uni.encode('windows-1251'), uni.encode('win-1251'))
- self.assertEqual(str.decode('windows-1251'), str.decode('win-1251'))
-
- text = u'\u8f6f\u4ef6\u540d\u79f0'
- str = uni.encode('gb2312')
- self.assertEqual(uni.encode('gb2312'), uni.encode('zh-cn'))
- self.assertEqual(str.decode('gb2312'), str.decode('zh-cn'))
-
-if __name__ == "__main__":
- unittest.main()
diff --git a/scrapy/tests/test_http_response.py b/scrapy/tests/test_http_response.py
index 3b0a144d2..081d4c864 100644
--- a/scrapy/tests/test_http_response.py
+++ b/scrapy/tests/test_http_response.py
@@ -2,6 +2,7 @@ import unittest
import weakref
from scrapy.http import Response, TextResponse, HtmlResponse, XmlResponse, Headers
+from scrapy.utils.encoding import resolve_encoding
from scrapy.conf import settings
@@ -112,10 +113,13 @@ class BaseResponseTest(unittest.TestCase):
body_str = body
assert isinstance(response.body, str)
- self.assertEqual(response.encoding, encoding)
+ self._assert_response_encoding(response, encoding)
self.assertEqual(response.body, body_str)
self.assertEqual(response.body_as_unicode(), body_unicode)
+ def _assert_response_encoding(self, response, encoding):
+ self.assertEqual(response.encoding, resolve_encoding(encoding))
+
class ResponseText(BaseResponseTest):
def test_no_unicode_url(self):
@@ -134,14 +138,14 @@ class TextResponseTest(BaseResponseTest):
assert isinstance(r2, self.response_class)
self.assertEqual(r2.url, "http://www.example.com/other")
- self.assertEqual(r2.encoding, "cp852")
+ self._assert_response_encoding(r2, "cp852")
self.assertEqual(r3.url, "http://www.example.com/other")
- self.assertEqual(r3.encoding, "latin1")
+ self.assertEqual(r3._declared_encoding(), "latin1")
def test_unicode_url(self):
# instantiate with unicode url without encoding (should set default encoding)
resp = self.response_class(u"http://www.example.com/")
- self.assertEqual(resp.encoding, settings['DEFAULT_RESPONSE_ENCODING'])
+ self._assert_response_encoding(resp, settings['DEFAULT_RESPONSE_ENCODING'])
# make sure urls are converted to str
resp = self.response_class(url=u"http://www.example.com/", encoding='utf-8')
@@ -175,14 +179,18 @@ class TextResponseTest(BaseResponseTest):
r2 = self.response_class("http://www.example.com", encoding='utf-8', body=u"\xa3")
r3 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=iso-8859-1"]}, body="\xa3")
r4 = self.response_class("http://www.example.com", body="\xa2\xa3")
+ r5 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=None"]}, body="\xc2\xa3")
- self.assertEqual(r1.headers_encoding(), "utf-8")
- self.assertEqual(r2.headers_encoding(), None)
- self.assertEqual(r2.encoding, 'utf-8')
- self.assertEqual(r3.headers_encoding(), "iso-8859-1")
- self.assertEqual(r3.encoding, 'iso-8859-1')
- self.assertEqual(r4.headers_encoding(), None)
- assert r4.body_encoding() is not None and r4.body_encoding() != 'ascii'
+ self.assertEqual(r1._headers_encoding(), "utf-8")
+ self.assertEqual(r2._headers_encoding(), None)
+ self.assertEqual(r2._declared_encoding(), 'utf-8')
+ self._assert_response_encoding(r2, 'utf-8')
+ self.assertEqual(r3._headers_encoding(), "iso-8859-1")
+ self.assertEqual(r3._declared_encoding(), "iso-8859-1")
+ self.assertEqual(r4._headers_encoding(), None)
+ self.assertEqual(r5._headers_encoding(), None)
+ self._assert_response_encoding(r5, "utf-8")
+ assert r4._body_inferred_encoding() is not None and r4._body_inferred_encoding() != 'ascii'
self._assert_response_values(r1, 'utf-8', u"\xa3")
self._assert_response_values(r2, 'utf-8', u"\xa3")
self._assert_response_values(r3, 'iso-8859-1', u"\xa3")
diff --git a/scrapy/tests/test_utils_encoding.py b/scrapy/tests/test_utils_encoding.py
new file mode 100644
index 000000000..b6800cd75
--- /dev/null
+++ b/scrapy/tests/test_utils_encoding.py
@@ -0,0 +1,25 @@
+import unittest
+
+from scrapy.utils.encoding import encoding_exists, resolve_encoding
+
+class UtilsEncodingTestCase(unittest.TestCase):
+
+ _ENCODING_ALIASES = {
+ 'foo': 'cp1252',
+ 'bar': 'none',
+ }
+
+ def test_resolve_encoding(self):
+ self.assertEqual(resolve_encoding('latin1', self._ENCODING_ALIASES),
+ 'latin1')
+ self.assertEqual(resolve_encoding('foo', self._ENCODING_ALIASES),
+ 'cp1252')
+
+ def test_encoding_exists(self):
+ assert encoding_exists('latin1', self._ENCODING_ALIASES)
+ assert encoding_exists('foo', self._ENCODING_ALIASES)
+ assert not encoding_exists('bar', self._ENCODING_ALIASES)
+ assert not encoding_exists('none', self._ENCODING_ALIASES)
+
+if __name__ == "__main__":
+ unittest.main()
diff --git a/scrapy/tests/test_utils_python.py b/scrapy/tests/test_utils_python.py
index 13d8be30c..dc46e1f91 100644
--- a/scrapy/tests/test_utils_python.py
+++ b/scrapy/tests/test_utils_python.py
@@ -1,7 +1,8 @@
+import operator
import unittest
from scrapy.utils.python import str_to_unicode, unicode_to_str, \
- memoizemethod_noargs, isbinarytext
+ memoizemethod_noargs, isbinarytext, equal_attributes
class UtilsPythonTestCase(unittest.TestCase):
def test_str_to_unicode(self):
@@ -61,5 +62,52 @@ class UtilsPythonTestCase(unittest.TestCase):
# finally some real binary bytes
assert isbinarytext("\x02\xa3")
+ def test_equal_attributes(self):
+ class Obj:
+ pass
+
+ a = Obj()
+ b = Obj()
+ # no attributes given return False
+ self.failIf(equal_attributes(a, b, []))
+ # not existent attributes
+ self.failIf(equal_attributes(a, b, ['x', 'y']))
+
+ a.x = 1
+ b.x = 1
+ # equal attribute
+ self.failUnless(equal_attributes(a, b, ['x']))
+
+ b.y = 2
+ # obj1 has no attribute y
+ self.failIf(equal_attributes(a, b, ['x', 'y']))
+
+ a.y = 2
+ # equal attributes
+ self.failUnless(equal_attributes(a, b, ['x', 'y']))
+
+ a.y = 1
+ # differente attributes
+ self.failIf(equal_attributes(a, b, ['x', 'y']))
+
+ # test callable
+ a.meta = {}
+ b.meta = {}
+ self.failUnless(equal_attributes(a, b, ['meta']))
+
+ # compare ['meta']['a']
+ a.meta['z'] = 1
+ b.meta['z'] = 1
+
+ get_z = operator.itemgetter('z')
+ get_meta = operator.attrgetter('meta')
+ compare_z = lambda obj: get_z(get_meta(obj))
+
+ self.failUnless(equal_attributes(a, b, [compare_z, 'x']))
+ # fail z equality
+ a.meta['z'] = 2
+ self.failIf(equal_attributes(a, b, [compare_z, 'x']))
+
+
if __name__ == "__main__":
unittest.main()
diff --git a/scrapy/tests/test_utils_response.py b/scrapy/tests/test_utils_response.py
index 9281f4f40..71c14569e 100644
--- a/scrapy/tests/test_utils_response.py
+++ b/scrapy/tests/test_utils_response.py
@@ -1,9 +1,10 @@
import unittest
+import urlparse
from scrapy.xlib.BeautifulSoup import BeautifulSoup
-from scrapy.http import Response, TextResponse
+from scrapy.http import Response, TextResponse, HtmlResponse
from scrapy.utils.response import body_or_str, get_base_url, get_meta_refresh, \
- response_httprepr, get_cached_beautifulsoup
+ response_httprepr, get_cached_beautifulsoup, open_in_browser
class ResponseUtilsTest(unittest.TestCase):
dummy_response = TextResponse(url='http://example.org/', body='dummy_response')
@@ -28,30 +29,46 @@ class ResponseUtilsTest(unittest.TestCase):
self.assertTrue(isinstance(body_or_str(u'text', unicode=True), unicode))
def test_get_base_url(self):
- response = Response(url='http://example.org', body="""\
+ response = HtmlResponse(url='https://example.org', body="""\
\
Dummy\
blahablsdfsal&\
""")
self.assertEqual(get_base_url(response), 'http://example.org/something')
+ # relative url with absolute path
+ response = HtmlResponse(url='https://example.org', body="""\
+ \
+ Dummy\
+ blahablsdfsal&\
+ """)
+ self.assertEqual(get_base_url(response), 'https://example.org/absolutepath')
+
+ # no scheme url
+ response = HtmlResponse(url='https://example.org', body="""\
+ \
+ Dummy\
+ blahablsdfsal&\
+ """)
+ self.assertEqual(get_base_url(response), 'https://noscheme.com/path')
+
def test_get_meta_refresh(self):
body = """
Dummy
blahablsdfsal&
"""
- response = Response(url='http://example.org', body=body)
+ response = TextResponse(url='http://example.org', body=body)
self.assertEqual(get_meta_refresh(response), (5, 'http://example.org/newpage'))
# refresh without url should return (None, None)
body = """"""
- response = Response(url='http://example.org', body=body)
+ response = TextResponse(url='http://example.org', body=body)
self.assertEqual(get_meta_refresh(response), (None, None))
body = """"""
- response = Response(url='http://example.org', body=body)
+ response = TextResponse(url='http://example.org', body=body)
self.assertEqual(get_meta_refresh(response), (5, 'http://example.org/newpage'))
# meta refresh in multiple lines
@@ -59,17 +76,17 @@ class ResponseUtilsTest(unittest.TestCase):
"""
- response = Response(url='http://example.org', body=body)
+ response = TextResponse(url='http://example.org', body=body)
self.assertEqual(get_meta_refresh(response), (1, 'http://example.org/newpage'))
# entities in the redirect url
body = """"""
- response = Response(url='http://example.com', body=body)
+ response = TextResponse(url='http://example.com', body=body)
self.assertEqual(get_meta_refresh(response), (3, 'http://www.example.com/other'))
# relative redirects
body = """"""
- response = Response(url='http://example.com/page/this.html', body=body)
+ response = TextResponse(url='http://example.com/page/this.html', body=body)
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/page/other.html'))
# non-standard encodings (utf-16)
@@ -80,7 +97,7 @@ class ResponseUtilsTest(unittest.TestCase):
# non-ascii chars in the url (default encoding - utf8)
body = """"""
- response = Response(url='http://example.com', body=body)
+ response = TextResponse(url='http://example.com', body=body)
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/to%C2%A3'))
# non-ascii chars in the url (custom encoding - latin1)
@@ -88,13 +105,8 @@ class ResponseUtilsTest(unittest.TestCase):
response = TextResponse(url='http://example.com', body=body, encoding='latin1')
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/to%C2%A3'))
- # wrong encodings (possibly caused by truncated chunks)
- body = """"""
- response = Response(url='http://example.com', body=body)
- self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/thisTHAT'))
-
# responses without refresh tag should return None None
- response = Response(url='http://example.org')
+ response = TextResponse(url='http://example.org')
self.assertEqual(get_meta_refresh(response), (None, None))
response = TextResponse(url='http://example.org')
self.assertEqual(get_meta_refresh(response), (None, None))
@@ -131,5 +143,18 @@ class ResponseUtilsTest(unittest.TestCase):
assert soup1 is soup2
assert soup1 is not soup3
+ def test_open_in_browser(self):
+ url = "http:///www.example.com/some/page.html"
+ body = " test page test body "
+ def browser_open(burl):
+ bbody = open(urlparse.urlparse(burl).path).read()
+ assert '' % url in bbody, " tag not added"
+ return True
+ response = HtmlResponse(url, body=body)
+ assert open_in_browser(response, _openfunc=browser_open), \
+ "Browser not called"
+ self.assertRaises(TypeError, open_in_browser, Response(url, body=body), \
+ debug=True)
+
if __name__ == "__main__":
unittest.main()
diff --git a/scrapy/utils/encoding.py b/scrapy/utils/encoding.py
index 9eb06d942..c7d645041 100644
--- a/scrapy/utils/encoding.py
+++ b/scrapy/utils/encoding.py
@@ -1,11 +1,20 @@
import codecs
-def add_encoding_alias(encoding, alias, overwrite=False):
+from scrapy.conf import settings
+
+_ENCODING_ALIASES = dict(settings['ENCODING_ALIASES_BASE'])
+_ENCODING_ALIASES.update(settings['ENCODING_ALIASES'])
+
+def encoding_exists(encoding, _aliases=_ENCODING_ALIASES):
+ """Returns ``True`` if encoding is valid, otherwise returns ``False``"""
try:
- codecs.lookup(alias)
- alias_exists = True
+ codecs.lookup(resolve_encoding(encoding, _aliases))
except LookupError:
- alias_exists = False
- if overwrite or not alias_exists:
- codec = codecs.lookup(encoding)
- codecs.register(lambda x: codec if x == alias else None)
+ return False
+ return True
+
+def resolve_encoding(alias, _aliases=_ENCODING_ALIASES):
+ """Return the encoding the given alias maps to, or the alias as passed if
+ no mapping is found.
+ """
+ return _aliases.get(alias.lower(), alias)
diff --git a/scrapy/utils/markup.py b/scrapy/utils/markup.py
index d422fbde5..5e2e0b1c1 100644
--- a/scrapy/utils/markup.py
+++ b/scrapy/utils/markup.py
@@ -60,10 +60,10 @@ def remove_entities(text, keep=(), remove_illegal=True, encoding='utf-8'):
return _ent_re.sub(convert_entity, str_to_unicode(text, encoding))
-def has_entities(text):
- return bool(_ent_re.search(str_to_unicode(text)))
+def has_entities(text, encoding=None):
+ return bool(_ent_re.search(str_to_unicode(text, encoding)))
-def replace_tags(text, token=''):
+def replace_tags(text, token='', encoding=None):
"""Replace all markup tags found in the given text by the given token. By
default token is a null string so it just remove all tags.
@@ -71,43 +71,44 @@ def replace_tags(text, token=''):
Always returns a unicode string.
"""
- return _tag_re.sub(token, str_to_unicode(text))
+ return _tag_re.sub(token, str_to_unicode(text, encoding))
-def remove_comments(text):
+def remove_comments(text, encoding=None):
""" Remove HTML Comments. """
- return re.sub('', u'', str_to_unicode(text), re.DOTALL)
+ return re.sub('', u'', str_to_unicode(text, encoding), re.DOTALL)
-def remove_tags(text, which_ones=()):
+def remove_tags(text, which_ones=(), encoding=None):
""" Remove HTML Tags only.
which_ones -- is a tuple of which tags we want to remove.
if is empty remove all tags.
"""
if which_ones:
- tags = ['<%s>|<%s .*?>|%s>' % (tag,tag,tag) for tag in which_ones]
+ tags = ['<%s>|<%s .*?>|%s>' % (tag, tag, tag) for tag in which_ones]
regex = '|'.join(tags)
else:
regex = '<.*?>'
retags = re.compile(regex, re.DOTALL | re.IGNORECASE)
- return retags.sub(u'', str_to_unicode(text))
+ return retags.sub(u'', str_to_unicode(text, encoding))
-def remove_tags_with_content(text, which_ones=()):
+def remove_tags_with_content(text, which_ones=(), encoding=None):
""" Remove tags and its content.
which_ones -- is a tuple of which tags with its content we want to remove.
if is empty do nothing.
"""
- text = str_to_unicode(text)
+ text = str_to_unicode(text, encoding)
if which_ones:
- tags = '|'.join(['<%s.*?%s>' % (tag,tag) for tag in which_ones])
+ tags = '|'.join(['<%s.*?%s>' % (tag, tag) for tag in which_ones])
retags = re.compile(tags, re.DOTALL | re.IGNORECASE)
text = retags.sub(u'', text)
return text
-def replace_escape_chars(text, which_ones=('\n','\t','\r'), replace_by=u''):
+def replace_escape_chars(text, which_ones=('\n', '\t', '\r'), replace_by=u'', \
+ encoding=None):
""" Remove escape chars. Default : \\n, \\t, \\r
which_ones -- is a tuple of which escape chars we want to remove.
@@ -117,10 +118,10 @@ def replace_escape_chars(text, which_ones=('\n','\t','\r'), replace_by=u''):
It defaults to '', so the escape chars are removed.
"""
for ec in which_ones:
- text = text.replace(ec, str_to_unicode(replace_by))
- return str_to_unicode(text)
+ text = text.replace(ec, str_to_unicode(replace_by, encoding))
+ return str_to_unicode(text, encoding)
-def unquote_markup(text, keep=(), remove_illegal=True):
+def unquote_markup(text, keep=(), remove_illegal=True, encoding=None):
"""
This function receives markup as a text (always a unicode string or a utf-8 encoded string) and does the following:
- removes entities (except the ones in 'keep') from any part of it that it's not inside a CDATA
@@ -138,7 +139,7 @@ def unquote_markup(text, keep=(), remove_illegal=True):
offset = match_e
yield txt[offset:]
- text = str_to_unicode(text)
+ text = str_to_unicode(text, encoding)
ret_text = u''
for fragment in _get_fragments(text, _cdata_re):
if isinstance(fragment, basestring):
diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py
index 7d43c4566..99aefcd74 100644
--- a/scrapy/utils/python.py
+++ b/scrapy/utils/python.py
@@ -64,13 +64,15 @@ def unique(list_, key=lambda x: x):
return result
-def str_to_unicode(text, encoding='utf-8'):
+def str_to_unicode(text, encoding=None):
"""Return the unicode representation of text in the given encoding. Unlike
.encode(encoding) this function can be applied directly to a unicode
object without the risk of double-decoding problems (which can happen if
you don't use the default 'ascii' encoding)
"""
+ if encoding is None:
+ encoding = 'utf-8'
if isinstance(text, str):
return text.decode(encoding)
elif isinstance(text, unicode):
@@ -78,13 +80,15 @@ def str_to_unicode(text, encoding='utf-8'):
else:
raise TypeError('str_to_unicode must receive a str or unicode object, got %s' % type(text).__name__)
-def unicode_to_str(text, encoding='utf-8'):
+def unicode_to_str(text, encoding=None):
"""Return the str representation of text in the given encoding. Unlike
.encode(encoding) this function can be applied directly to a str
object without the risk of double-decoding problems (which can happen if
you don't use the default 'ascii' encoding)
"""
+ if encoding is None:
+ encoding = 'utf-8'
if isinstance(text, unicode):
return text.encode(encoding)
elif isinstance(text, str):
@@ -216,3 +220,28 @@ def get_func_args(func):
else:
raise TypeError('%s is not callable' % type(func))
return func_args
+
+
+def equal_attributes(obj1, obj2, attributes):
+ """Compare two objects attributes"""
+ # not attributes given return False by default
+ if not attributes:
+ return False
+
+ for attr in attributes:
+ # support callables like itemgetter
+ if callable(attr):
+ if not attr(obj1) == attr(obj2):
+ return False
+ else:
+ # check that objects has attribute
+ if not hasattr(obj1, attr):
+ return False
+ if not hasattr(obj2, attr):
+ return False
+ # compare object attributes
+ if not getattr(obj1, attr) == getattr(obj2, attr):
+ return False
+ # all attributes equal
+ return True
+
diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py
index 6ed6f43a3..ca258ef44 100644
--- a/scrapy/utils/response.py
+++ b/scrapy/utils/response.py
@@ -18,7 +18,8 @@ from scrapy.xlib.BeautifulSoup import BeautifulSoup
from scrapy.http import Response, HtmlResponse
def body_or_str(obj, unicode=True):
- assert isinstance(obj, (Response, basestring)), "obj must be Response or basestring, not %s" % type(obj).__name__
+ assert isinstance(obj, (Response, basestring)), \
+ "obj must be Response or basestring, not %s" % type(obj).__name__
if isinstance(obj, Response):
return obj.body_as_unicode() if unicode else obj.body
elif isinstance(obj, str):
@@ -26,16 +27,17 @@ def body_or_str(obj, unicode=True):
else:
return obj if unicode else obj.encode('utf-8')
-BASEURL_RE = re.compile(r']*http-equiv[^>]*refresh[^>]*content\s*=\s*(?P["\'])(?P\d+)\s*;\s*url=(?P.*?)(?P=quote)', re.DOTALL | re.IGNORECASE)
+META_REFRESH_RE = re.compile(ur']*http-equiv[^>]*refresh[^>]*content\s*=\s*(?P["\'])(?P\d+)\s*;\s*url=(?P.*?)(?P=quote)', \
+ re.DOTALL | re.IGNORECASE)
_metaref_cache = weakref.WeakKeyDictionary()
def get_meta_refresh(response):
"""Parse the http-equiv parameter of the HTML meta element from the given
@@ -46,9 +48,7 @@ def get_meta_refresh(response):
If no meta redirect is found, (None, None) is returned.
"""
if response not in _metaref_cache:
- encoding = getattr(response, 'encoding', None) or 'utf-8'
- body_chunk = remove_entities(unicode(response.body[0:4096], encoding, \
- errors='ignore'))
+ body_chunk = remove_entities(response.body_as_unicode()[0:4096])
match = META_REFRESH_RE.search(body_chunk)
if match:
interval = int(match.group('int'))
@@ -92,7 +92,7 @@ def response_httprepr(response):
s += response.body
return s
-def open_in_browser(response):
+def open_in_browser(response, _openfunc=webbrowser.open):
"""Open the given response in a local web browser, populating the
tag for external links to work
"""
@@ -106,4 +106,4 @@ def open_in_browser(response):
fd, fname = tempfile.mkstemp('.html')
os.write(fd, body)
os.close(fd)
- webbrowser.open("file://%s" % fname)
+ return _openfunc("file://%s" % fname)
diff --git a/scrapy/utils/url.py b/scrapy/utils/url.py
index 1c2fe18ae..7496328fc 100644
--- a/scrapy/utils/url.py
+++ b/scrapy/utils/url.py
@@ -130,7 +130,8 @@ def add_or_replace_parameter(url, name, new_value, sep='&', url_is_quoted=False)
name+'='+new_value)
return next_url
-def canonicalize_url(url, keep_blank_values=True, keep_fragments=False):
+def canonicalize_url(url, keep_blank_values=True, keep_fragments=False, \
+ encoding=None):
"""Canonicalize the given url by applying the following procedures:
- sort query arguments, first by key, then by value
@@ -147,7 +148,7 @@ def canonicalize_url(url, keep_blank_values=True, keep_fragments=False):
For examples see the tests in scrapy.tests.test_utils_url
"""
- url = unicode_to_str(url)
+ url = unicode_to_str(url, encoding)
scheme, netloc, path, params, query, fragment = urlparse.urlparse(url)
keyvals = cgi.parse_qsl(query, keep_blank_values)
keyvals.sort()