mirror of https://github.com/scrapy/scrapy.git
Automated merge with http://hg.scrapy.org/scrapy-0.8
This commit is contained in:
commit
2ed8a5bfb5
39
AUTHORS
39
AUTHORS
|
|
@ -1,28 +1,25 @@
|
|||
Scrapy was brought to life by Shane Evans while hacking a scraping framework
|
||||
prototype for Mydeco (mydeco.com). It soon became maintained, extended and
|
||||
improved by Insophia (insophia.com), with the sponsorship of By Design (the
|
||||
company behind Mydeco).
|
||||
improved by Insophia (insophia.com), with the initial sponsorship of Mydeco to
|
||||
bootstrap the project.
|
||||
|
||||
Here is the list of the primary authors & contributors, along with their user
|
||||
name (in Scrapy trac/subversion). Emails are intentionally left out to avoid
|
||||
spam.
|
||||
Here is the list of the primary authors & contributors:
|
||||
|
||||
* Pablo Hoffman (pablo)
|
||||
* Daniel Graña (daniel)
|
||||
* Martin Olveyra (olveyra)
|
||||
* Gabriel García (elpolilla)
|
||||
* Michael Cetrulo (samus_)
|
||||
* Artem Bogomyagkov (artem)
|
||||
* Damian Canabal (calarval)
|
||||
* Andres Moreira (andres)
|
||||
* Ismael Carnales (ismael)
|
||||
* Matías Aguirre (omab)
|
||||
* German Hoffman (german)
|
||||
* Anibal Pacheco (anibal)
|
||||
* Pablo Hoffman
|
||||
* Daniel Graña
|
||||
* Martin Olveyra
|
||||
* Gabriel García
|
||||
* Michael Cetrulo
|
||||
* Artem Bogomyagkov
|
||||
* Damian Canabal
|
||||
* Andres Moreira
|
||||
* Ismael Carnales
|
||||
* Matías Aguirre
|
||||
* German Hoffmann
|
||||
* Anibal Pacheco
|
||||
* Bruno Deferrari
|
||||
* Shane Evans
|
||||
|
||||
And here is the list of people who have helped to put the Scrapy homepage live:
|
||||
|
||||
* Ezequiel Rivero (ezequiel)
|
||||
* Ezequiel Rivero
|
||||
* Patrick Mezard
|
||||
* Rolando Espinoza
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,128 @@
|
|||
.. _topics-crawlspider-v2:
|
||||
|
||||
==============
|
||||
CrawlSpider v2
|
||||
==============
|
||||
|
||||
Introduction
|
||||
============
|
||||
|
||||
TODO: introduction
|
||||
|
||||
Rules Matching
|
||||
==============
|
||||
|
||||
TODO: describe purpose of rules
|
||||
|
||||
Request Extractors & Processors
|
||||
===============================
|
||||
|
||||
TODO: describe purpose of extractors & processors
|
||||
|
||||
Examples
|
||||
========
|
||||
|
||||
TODO: plenty of examples
|
||||
|
||||
|
||||
.. module:: scrapy.contrib_exp.crawlspider.spider
|
||||
:synopsis: CrawlSpider
|
||||
|
||||
|
||||
Reference
|
||||
=========
|
||||
|
||||
CrawlSpider
|
||||
-----------
|
||||
|
||||
TODO: describe crawlspider
|
||||
|
||||
.. class:: CrawlSpider
|
||||
|
||||
TODO: describe class
|
||||
|
||||
|
||||
.. module:: scrapy.contrib_exp.crawlspider.rules
|
||||
:synopsis: Rules
|
||||
|
||||
Rules
|
||||
-----
|
||||
|
||||
TODO: describe spider rules
|
||||
|
||||
.. class:: Rule
|
||||
|
||||
TODO: describe Rules class
|
||||
|
||||
|
||||
.. module:: scrapy.contrib_exp.crawlspider.reqext
|
||||
:synopsis: Request Extractors
|
||||
|
||||
Request Extractors
|
||||
------------------
|
||||
|
||||
TODO: describe extractors purpose
|
||||
|
||||
.. class:: BaseSgmlRequestExtractor
|
||||
|
||||
TODO: describe base extractor
|
||||
|
||||
.. class:: SgmlRequestExtractor
|
||||
|
||||
TODO: describe sgml extractor
|
||||
|
||||
.. class:: XPathRequestExtractor
|
||||
|
||||
TODO: describe xpath request extractor
|
||||
|
||||
|
||||
.. module:: scrapy.contrib_exp.crawlspider.reqproc
|
||||
:synopsis: Request Processors
|
||||
|
||||
Request Processors
|
||||
------------------
|
||||
|
||||
TODO: describe request processors
|
||||
|
||||
.. class:: Canonicalize
|
||||
|
||||
TODO: describe proc
|
||||
|
||||
.. class:: Unique
|
||||
|
||||
TODO: describe unique
|
||||
|
||||
.. class:: FilterDomain
|
||||
|
||||
TODO: describe filter domain
|
||||
|
||||
.. class:: FilterUrl
|
||||
|
||||
TODO: describe filter url
|
||||
|
||||
|
||||
.. module:: scrapy.contrib_exp.crawlspider.matchers
|
||||
:synopsis: Matchers
|
||||
|
||||
Request/Response Matchers
|
||||
-------------------------
|
||||
|
||||
TODO: describe matchers
|
||||
|
||||
.. class:: BaseMatcher
|
||||
|
||||
TODO: describe base matcher
|
||||
|
||||
.. class:: UrlMatcher
|
||||
|
||||
TODO: describe url matcher
|
||||
|
||||
.. class:: UrlRegexMatcher
|
||||
|
||||
TODO: describe UrlListMatcher
|
||||
|
||||
.. class:: UrlListMatcher
|
||||
|
||||
TODO: describe url list matcher
|
||||
|
||||
|
||||
|
|
@ -21,3 +21,4 @@ it's properly merged) . Use at your own risk.
|
|||
|
||||
djangoitems
|
||||
scheduler-middleware
|
||||
crawlspider-v2
|
||||
|
|
|
|||
|
|
@ -420,8 +420,8 @@ scraped so far, the code for our Spider should be like this::
|
|||
|
||||
Now doing a crawl on the dmoz.org domain yields ``DmozItem``'s::
|
||||
|
||||
[dmoz.org] DEBUG: Scraped DmozItem({'title': [u'Text Processing in Python'], 'link': [u'http://gnosis.cx/TPiP/'], 'desc': [u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n']}) in <http://www.dmoz.org/Computers/Programming/Languages/Python/Books/>
|
||||
[dmoz.org] DEBUG: Scraped DmozItem({'title': [u'XML Processing with Python'], 'link': [u'http://www.informit.com/store/product.aspx?isbn=0130211192'], 'desc': [u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n']}) in <http://www.dmoz.org/Computers/Programming/Languages/Python/Books/>
|
||||
[dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By David Mertz; Addison Wesley. Book in progress, full text, ASCII format. Asks for feedback. [author website, Gnosis Software, Inc.]\n'], link=[u'http://gnosis.cx/TPiP/'], title=[u'Text Processing in Python']) in <http://www.dmoz.org/Computers/Programming/Languages/Python/Books/>
|
||||
[dmoz.org] DEBUG: Scraped DmozItem(desc=[u' - By Sean McGrath; Prentice Hall PTR, 2000, ISBN 0130211192, has CD-ROM. Methods to build XML applications fast, Python tutorial, DOM and SAX, new Pyxie open source XML processing library. [Prentice Hall PTR]\n'], link=[u'http://www.informit.com/store/product.aspx?isbn=0130211192'], title=[u'XML Processing with Python']) in <http://www.dmoz.org/Computers/Programming/Languages/Python/Books/>
|
||||
|
||||
|
||||
Storing the data (using an Item Pipeline)
|
||||
|
|
|
|||
|
|
@ -5,14 +5,17 @@ Sending email
|
|||
=============
|
||||
|
||||
.. module:: scrapy.mail
|
||||
:synopsis: Helpers to easily send e-mail.
|
||||
:synopsis: Email sending facility
|
||||
|
||||
Although Python makes sending e-mail relatively easy via the `smtplib`_
|
||||
library, Scrapy provides its own class for sending emails which is very easy to
|
||||
use and it's implemented using `Twisted non-blocking IO`_, to avoid affecting
|
||||
the crawling performance.
|
||||
library, Scrapy provides its own facility for sending emails which is very easy
|
||||
to use and it's implemented using `Twisted non-blocking IO`_, to avoid
|
||||
interfering with the non-blocking IO of the crawler.
|
||||
|
||||
It's also very easy to configure by just configuring a few settings.
|
||||
|
||||
.. _smtplib: http://docs.python.org/library/smtplib.html
|
||||
.. _Twisted non-blocking IO: http://twistedmatrix.com/projects/core/documentation/howto/async.html
|
||||
|
||||
It also has built-in support for sending attachments.
|
||||
|
||||
|
|
@ -34,30 +37,45 @@ uses `Twisted non-blocking IO`_, like the rest of the framework.
|
|||
|
||||
.. class:: MailSender(smtphost, mailfrom)
|
||||
|
||||
``smtphost`` is a string with the SMTP host to use for sending the emails.
|
||||
If omitted, :setting:`MAIL_HOST` will be used.
|
||||
:param smtphost: the SMTP host to use for sending the emails. If omitted,
|
||||
:setting:`MAIL_HOST` setting will be used.
|
||||
:type smtphost: str
|
||||
|
||||
``mailfrom`` is a string with the email address to use for sending messages
|
||||
(in the ``From:`` header). If omitted, :setting:`MAIL_FROM` will be used.
|
||||
:param mailfrom: the address used to send emails (in the ``From:`` header).
|
||||
If omitted, :setting:`MAIL_FROM` setting will be used.
|
||||
:type mailfrom: str
|
||||
|
||||
.. method:: MailSender.send(to, subject, body, cc=None, attachs=())
|
||||
.. method:: send(to, subject, body, cc=None, attachs=())
|
||||
|
||||
Send mail to the given recipients
|
||||
Send email to the given recipients
|
||||
|
||||
``to`` is a list of email recipients
|
||||
:param to: the email recipients
|
||||
:type to: list
|
||||
|
||||
``subject`` is a string with the subject of the message
|
||||
:param subject: the subject of the email
|
||||
:type subject: str
|
||||
|
||||
``cc`` is a list of emails to CC
|
||||
:param cc: the emails to CC
|
||||
:type cc: list
|
||||
|
||||
``body`` is a string with the body of the message
|
||||
:param body: the email body
|
||||
:type body: str
|
||||
|
||||
``attachs`` is an iterable of tuples (attach_name, mimetype, file_object)
|
||||
where:
|
||||
|
||||
``attach_name`` is a string with the name will appear on the emails attachment
|
||||
``mimetype`` is the mimetype of the attachment
|
||||
``file_object`` is a readable file object
|
||||
:param attachs: an iterable of tuples ``(attach_name, mimetype,
|
||||
file_object)`` where ``attach_name`` is a string with the name will
|
||||
appear on the emails attachment, ``mimetype`` is the mimetype of the
|
||||
attachment and ``file_object`` is a readable file object with the
|
||||
contents of the attachment
|
||||
:type attachs: iterable
|
||||
|
||||
|
||||
.. _Twisted non-blocking IO: http://twistedmatrix.com/projects/core/documentation/howto/async.html
|
||||
MailSender settings
|
||||
===================
|
||||
|
||||
These settings define the default constructor values of the :class:`MailSender`
|
||||
class, and can be used to configure email notifications in your project without
|
||||
writing any code (for those extensions that use the :class:`MailSender` class):
|
||||
|
||||
* :setting:`MAIL_FROM`
|
||||
* :setting:`MAIL_HOST`
|
||||
|
||||
|
|
|
|||
|
|
@ -98,10 +98,10 @@ spider returns multiples items with the same id::
|
|||
del self.duplicates[spider]
|
||||
|
||||
def process_item(self, spider, item):
|
||||
if item.id in self.duplicates[spider]:
|
||||
if item['id'] in self.duplicates[spider]:
|
||||
raise DropItem("Duplicate item found: %s" % item)
|
||||
else:
|
||||
self.duplicates[spider].add(item.id)
|
||||
self.duplicates[spider].add(item['id'])
|
||||
return item
|
||||
|
||||
Built-in Item Pipelines reference
|
||||
|
|
@ -178,7 +178,7 @@ on the respective Item Exporter to get more info.
|
|||
the first line. This format requires you to specify the fields to export
|
||||
using the :setting:`EXPORT_FIELDS` setting.
|
||||
|
||||
* ``jsonlines``: uses a :class:`~jsonlines.JsonLinesItemExporter`
|
||||
* ``json``: uses a :class:`~jsonlines.JsonLinesItemExporter`
|
||||
|
||||
* ``pickle``: uses a :class:`PickleItemExporter`
|
||||
|
||||
|
|
|
|||
|
|
@ -129,3 +129,14 @@ scrapy.log module
|
|||
|
||||
Log level for debugging messages (recommended level for development)
|
||||
|
||||
Logging settings
|
||||
================
|
||||
|
||||
These settings can be used to configure the logging:
|
||||
|
||||
* :setting:`LOG_ENABLED`
|
||||
* :setting:`LOG_ENCODING`
|
||||
* :setting:`LOG_FILE`
|
||||
* :setting:`LOG_LEVEL`
|
||||
* :setting:`LOG_STDOUT`
|
||||
|
||||
|
|
|
|||
|
|
@ -466,12 +466,14 @@ TextResponse objects
|
|||
|
||||
.. attribute:: TextResponse.encoding
|
||||
|
||||
A string with the encoding of this response. The encoding is resolved in the
|
||||
following order:
|
||||
A string with the encoding of this response. The encoding is resolved by
|
||||
trying the following mechanisms, in order:
|
||||
|
||||
1. the encoding passed in the constructor `encoding` argument
|
||||
|
||||
2. the encoding declared in the Content-Type HTTP header
|
||||
2. the encoding declared in the Content-Type HTTP header. If this
|
||||
encoding is not valid (ie. unknown), it is ignored and the next
|
||||
resolution mechanism is tried.
|
||||
|
||||
3. the encoding declared in the response body. The TextResponse class
|
||||
doesn't provide any special functionality for this. However, the
|
||||
|
|
@ -483,23 +485,11 @@ TextResponse objects
|
|||
:class:`TextResponse` objects support the following methods in addition to
|
||||
the standard :class:`Response` ones:
|
||||
|
||||
.. method:: TextResponse.headers_encoding()
|
||||
|
||||
Returns a string with the encoding declared in the headers (ie. the
|
||||
Content-Type HTTP header).
|
||||
|
||||
.. method:: TextResponse.body_encoding()
|
||||
|
||||
Returns a string with the encoding of the body, either declared or inferred
|
||||
from its contents. The body encoding declaration is implemented in
|
||||
:class:`TextResponse` subclasses such as: :class:`HtmlResponse` or
|
||||
:class:`XmlResponse`.
|
||||
|
||||
.. method:: TextResponse.body_as_unicode()
|
||||
|
||||
Returns the body of the response as unicode. This is equivalent to::
|
||||
|
||||
response.body.encode(response.encoding)
|
||||
response.body.decode(response.encoding)
|
||||
|
||||
But **not** equivalent to::
|
||||
|
||||
|
|
|
|||
|
|
@ -340,16 +340,6 @@ Default: ``True``
|
|||
|
||||
Whether to collect depth stats.
|
||||
|
||||
.. setting:: DOMAIN_SCHEDULER
|
||||
|
||||
SPIDER_SCHEDULER
|
||||
----------------
|
||||
|
||||
Default: ``'scrapy.contrib.spiderscheduler.FifoSpiderScheduler'``
|
||||
|
||||
The Spider Scheduler to use. The spider scheduler returns the next spider to
|
||||
scrape.
|
||||
|
||||
.. setting:: DOWNLOADER_DEBUG
|
||||
|
||||
DOWNLOADER_DEBUG
|
||||
|
|
@ -418,6 +408,15 @@ supported. Example::
|
|||
|
||||
DOWNLOAD_DELAY = 0.25 # 250 ms of delay
|
||||
|
||||
This setting is also affected by the :setting:`RANDOMIZE_DOWNLOAD_DELAY`
|
||||
setting (which is enabled by default). By default, Scrapy doesn't wait a fixed
|
||||
amount of time between requests, but uses a random interval between 0.5 and 1.5
|
||||
* :setting:`DOWNLOAD_DELAY`.
|
||||
|
||||
Another way to change the download delay (per spider, instead of globally) is
|
||||
by using the ``download_delay`` spider attribute, which takes more precedence
|
||||
than this setting.
|
||||
|
||||
.. setting:: DOWNLOAD_TIMEOUT
|
||||
|
||||
DOWNLOAD_TIMEOUT
|
||||
|
|
@ -439,6 +438,52 @@ The class used to detect and filter duplicate requests.
|
|||
The default (``RequestFingerprintDupeFilter``) filters based on request fingerprint
|
||||
(using ``scrapy.utils.request.request_fingerprint``) and grouping per domain.
|
||||
|
||||
.. setting:: ENCODING_ALIASES
|
||||
|
||||
ENCODING_ALIASES
|
||||
----------------
|
||||
|
||||
Default: ``{}``
|
||||
|
||||
A mapping of custom encoding aliases for your project, where the keys are the
|
||||
aliases (and must be lower case) and the values are the encodings they map to.
|
||||
|
||||
This setting extends the :setting:`ENCODING_ALIASES_BASE` setting which
|
||||
contains some default mappings.
|
||||
|
||||
.. setting:: ENCODING_ALIASES_BASE
|
||||
|
||||
ENCODING_ALIASES_BASE
|
||||
---------------------
|
||||
|
||||
Default::
|
||||
|
||||
{
|
||||
'gb2312': 'zh-cn',
|
||||
'cp1251': 'win-1251',
|
||||
'macintosh' : 'mac-roman',
|
||||
'x-sjis': 'shift-jis',
|
||||
'iso-8859-1': 'cp1252',
|
||||
'iso8859-1': 'cp1252',
|
||||
'8859': 'cp1252',
|
||||
'cp819': 'cp1252',
|
||||
'latin': 'cp1252',
|
||||
'latin1': 'cp1252',
|
||||
'latin_1': 'cp1252',
|
||||
'l1': 'cp1252',
|
||||
}
|
||||
|
||||
The default encoding aliases defined in Scrapy. Don't override this setting in
|
||||
your project, override :setting:`ENCODING_ALIASES` instead.
|
||||
|
||||
The reason why `ISO-8859-1`_ (and all its aliases) are mapped to `CP1252`_ is
|
||||
due to a well known browser hack. For more information see: `Character
|
||||
encodings in HTML`_.
|
||||
|
||||
.. _ISO-8859-1: http://en.wikipedia.org/wiki/ISO/IEC_8859-1
|
||||
.. _CP1252: http://en.wikipedia.org/wiki/Windows-1252
|
||||
.. _Character encodings in HTML: http://en.wikipedia.org/wiki/Character_encodings_in_HTML
|
||||
|
||||
.. setting:: EXTENSIONS
|
||||
|
||||
EXTENSIONS
|
||||
|
|
@ -517,7 +562,16 @@ LOG_ENABLED
|
|||
|
||||
Default: ``True``
|
||||
|
||||
Enable logging.
|
||||
Whether to enable logging.
|
||||
|
||||
.. setting:: LOG_ENCODING
|
||||
|
||||
LOG_ENCODING
|
||||
------------
|
||||
|
||||
Default: ``'utf-8'``
|
||||
|
||||
The encoding to use for logging.
|
||||
|
||||
.. setting:: LOG_FILE
|
||||
|
||||
|
|
@ -677,6 +731,27 @@ Example::
|
|||
|
||||
NEWSPIDER_MODULE = 'mybot.spiders_dev'
|
||||
|
||||
.. setting:: RANDOMIZE_DOWNLOAD_DELAY
|
||||
|
||||
RANDOMIZE_DOWNLOAD_DELAY
|
||||
------------------------
|
||||
|
||||
Default: ``True``
|
||||
|
||||
If enabled, Scrapy will wait a random amount of time (between 0.5 and 1.5
|
||||
* :setting:`DOWNLOAD_DELAY`) while fetching requests from the same
|
||||
spider.
|
||||
|
||||
This randomization decreases the chance of the crawler being detected (and
|
||||
subsequently blocked) by sites which analyze requests looking for statistically
|
||||
significant similarities in the time between their times.
|
||||
|
||||
The randomization policy is the same used by `wget`_ ``--random-wait`` option.
|
||||
|
||||
If :setting:`DOWNLOAD_DELAY` is zero (default) this option has no effect.
|
||||
|
||||
.. _wget: http://www.gnu.org/software/wget/manual/wget.html
|
||||
|
||||
.. setting:: REDIRECT_MAX_TIMES
|
||||
|
||||
REDIRECT_MAX_TIMES
|
||||
|
|
@ -773,7 +848,7 @@ The scheduler to use for crawling.
|
|||
SCHEDULER_ORDER
|
||||
---------------
|
||||
|
||||
Default: ``'BFO'``
|
||||
Default: ``'DFO'``
|
||||
|
||||
Scope: ``scrapy.core.scheduler``
|
||||
|
||||
|
|
@ -858,6 +933,16 @@ Example::
|
|||
|
||||
SPIDER_MODULES = ['mybot.spiders_prod', 'mybot.spiders_dev']
|
||||
|
||||
.. setting:: SPIDER_SCHEDULER
|
||||
|
||||
SPIDER_SCHEDULER
|
||||
----------------
|
||||
|
||||
Default: ``'scrapy.contrib.spiderscheduler.FifoSpiderScheduler'``
|
||||
|
||||
The Spider Scheduler to use. The spider scheduler returns the next spider to
|
||||
scrape.
|
||||
|
||||
.. setting:: STATS_CLASS
|
||||
|
||||
STATS_CLASS
|
||||
|
|
|
|||
|
|
@ -0,0 +1 @@
|
|||
# googledir project
|
||||
|
|
@ -0,0 +1,16 @@
|
|||
# Define here the models for your scraped items
|
||||
#
|
||||
# See documentation in:
|
||||
# http://doc.scrapy.org/topics/items.html
|
||||
|
||||
from scrapy.item import Item, Field
|
||||
|
||||
class GoogledirItem(Item):
|
||||
|
||||
name = Field(default='')
|
||||
url = Field(default='')
|
||||
description = Field(default='')
|
||||
|
||||
def __str__(self):
|
||||
return "Google Category: name=%s url=%s" \
|
||||
% (self['name'], self['url'])
|
||||
|
|
@ -0,0 +1,22 @@
|
|||
# Define your item pipelines here
|
||||
#
|
||||
# Don't forget to add your pipeline to the ITEM_PIPELINES setting
|
||||
# See: http://doc.scrapy.org/topics/item-pipeline.html
|
||||
|
||||
from scrapy.core.exceptions import DropItem
|
||||
|
||||
class FilterWordsPipeline(object):
|
||||
"""
|
||||
A pipeline for filtering out items which contain certain
|
||||
words in their description
|
||||
"""
|
||||
|
||||
# put all words in lowercase
|
||||
words_to_filter = ['politics', 'religion']
|
||||
|
||||
def process_item(self, spider, item):
|
||||
for word in self.words_to_filter:
|
||||
if word in unicode(item['description']).lower():
|
||||
raise DropItem("Contains forbidden word: %s" % word)
|
||||
else:
|
||||
return item
|
||||
|
|
@ -0,0 +1,21 @@
|
|||
# Scrapy settings for googledir project
|
||||
#
|
||||
# For simplicity, this file contains only the most important settings by
|
||||
# default. All the other settings are documented here:
|
||||
#
|
||||
# http://doc.scrapy.org/topics/settings.html
|
||||
#
|
||||
# Or you can copy and paste them from where they're defined in Scrapy:
|
||||
#
|
||||
# scrapy/conf/default_settings.py
|
||||
#
|
||||
|
||||
BOT_NAME = 'googledir'
|
||||
BOT_VERSION = '1.0'
|
||||
|
||||
SPIDER_MODULES = ['googledir.spiders']
|
||||
NEWSPIDER_MODULE = 'googledir.spiders'
|
||||
DEFAULT_ITEM_CLASS = 'googledir.items.GoogledirItem'
|
||||
USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION)
|
||||
|
||||
ITEM_PIPELINES = ['googledir.pipelines.FilterWordsPipeline']
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
# This package will contain the spiders of your Scrapy project
|
||||
#
|
||||
# To create the first spider for your project use this command:
|
||||
#
|
||||
# scrapy-ctl.py genspider myspider myspider-domain.com
|
||||
#
|
||||
# For more info see:
|
||||
# http://doc.scrapy.org/topics/spiders.html
|
||||
|
|
@ -0,0 +1,40 @@
|
|||
from scrapy.selector import HtmlXPathSelector
|
||||
from scrapy.contrib.loader import XPathItemLoader
|
||||
from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule
|
||||
|
||||
from googledir.items import GoogledirItem
|
||||
|
||||
class GoogleDirectorySpider(CrawlSpider):
|
||||
|
||||
domain_name = 'directory.google.com'
|
||||
start_urls = ['http://directory.google.com/']
|
||||
|
||||
rules = (
|
||||
# search for categories pattern and follow links
|
||||
Rule(r'/[A-Z][a-zA-Z_/]+$', 'parse_category', follow=True),
|
||||
)
|
||||
|
||||
def parse_category(self, response):
|
||||
# The main selector we're using to extract data from the page
|
||||
main_selector = HtmlXPathSelector(response)
|
||||
|
||||
# The XPath to website links in the directory page
|
||||
xpath = '//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td/font'
|
||||
|
||||
# Get a list of (sub) selectors to each website node pointed by the XPath
|
||||
sub_selectors = main_selector.select(xpath)
|
||||
|
||||
# Iterate over the sub-selectors to extract data for each website
|
||||
for selector in sub_selectors:
|
||||
item = GoogledirItem()
|
||||
|
||||
l = XPathItemLoader(item=item, selector=selector)
|
||||
l.add_xpath('name', 'a/text()')
|
||||
l.add_xpath('url', 'a/@href')
|
||||
l.add_xpath('description', 'font[2]/text()')
|
||||
|
||||
# Here we populate the item and yield it
|
||||
yield l.load_item()
|
||||
|
||||
SPIDER = GoogleDirectorySpider()
|
||||
|
||||
|
|
@ -0,0 +1,7 @@
|
|||
#!/usr/bin/env python
|
||||
|
||||
import os
|
||||
os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'googledir.settings')
|
||||
|
||||
from scrapy.command.cmdline import execute
|
||||
execute()
|
||||
|
|
@ -0,0 +1 @@
|
|||
# package
|
||||
|
|
@ -0,0 +1,12 @@
|
|||
# Define here the models for your scraped items
|
||||
#
|
||||
# See documentation in:
|
||||
# http://doc.scrapy.org/topics/items.html
|
||||
|
||||
from scrapy.item import Item, Field
|
||||
|
||||
class ImdbItem(Item):
|
||||
# define the fields for your item here like:
|
||||
# name = Field()
|
||||
title = Field()
|
||||
url = Field()
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
# Define your item pipelines here
|
||||
#
|
||||
# Don't forget to add your pipeline to the ITEM_PIPELINES setting
|
||||
# See: http://doc.scrapy.org/topics/item-pipeline.html
|
||||
|
||||
class ImdbPipeline(object):
|
||||
def process_item(self, spider, item):
|
||||
return item
|
||||
|
|
@ -0,0 +1,20 @@
|
|||
# Scrapy settings for imdb project
|
||||
#
|
||||
# For simplicity, this file contains only the most important settings by
|
||||
# default. All the other settings are documented here:
|
||||
#
|
||||
# http://doc.scrapy.org/topics/settings.html
|
||||
#
|
||||
# Or you can copy and paste them from where they're defined in Scrapy:
|
||||
#
|
||||
# scrapy/conf/default_settings.py
|
||||
#
|
||||
|
||||
BOT_NAME = 'imdb'
|
||||
BOT_VERSION = '1.0'
|
||||
|
||||
SPIDER_MODULES = ['imdb.spiders']
|
||||
NEWSPIDER_MODULE = 'imdb.spiders'
|
||||
DEFAULT_ITEM_CLASS = 'imdb.items.ImdbItem'
|
||||
USER_AGENT = '%s/%s' % (BOT_NAME, BOT_VERSION)
|
||||
|
||||
|
|
@ -0,0 +1,8 @@
|
|||
# This package will contain the spiders of your Scrapy project
|
||||
#
|
||||
# To create the first spider for your project use this command:
|
||||
#
|
||||
# scrapy-ctl.py genspider myspider myspider-domain.com
|
||||
#
|
||||
# For more info see:
|
||||
# http://doc.scrapy.org/topics/spiders.html
|
||||
|
|
@ -0,0 +1,140 @@
|
|||
from scrapy.http import Request
|
||||
from scrapy.selector import HtmlXPathSelector
|
||||
from scrapy.contrib.loader import XPathItemLoader
|
||||
from scrapy.contrib_exp.crawlspider import CrawlSpider, Rule
|
||||
from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize, \
|
||||
FilterDupes, FilterUrl
|
||||
from scrapy.utils.url import urljoin_rfc
|
||||
|
||||
from imdb.items import ImdbItem, Field
|
||||
|
||||
from itertools import chain, imap, izip
|
||||
|
||||
class UsaOpeningWeekMovie(ImdbItem):
|
||||
pass
|
||||
|
||||
class UsaTopWeekMovie(ImdbItem):
|
||||
pass
|
||||
|
||||
class Top250Movie(ImdbItem):
|
||||
rank = Field()
|
||||
rating = Field()
|
||||
year = Field()
|
||||
votes = Field()
|
||||
|
||||
class MovieItem(ImdbItem):
|
||||
release_date = Field()
|
||||
tagline = Field()
|
||||
|
||||
|
||||
class ImdbSiteSpider(CrawlSpider):
|
||||
domain_name = 'imdb.com'
|
||||
start_urls = ['http://www.imdb.com/']
|
||||
|
||||
# extract requests using this classes from urls matching 'follow' flag
|
||||
request_extractors = [
|
||||
SgmlRequestExtractor(tags=['a'], attrs=['href']),
|
||||
]
|
||||
|
||||
# process requests using this classes from urls matching 'follow' flag
|
||||
request_processors = [
|
||||
Canonicalize(),
|
||||
FilterDupes(),
|
||||
FilterUrl(deny=r'/tt\d+/$'), # deny movie url as we will dispatch
|
||||
# manually the movie requests
|
||||
]
|
||||
|
||||
# include domain bit for demo purposes
|
||||
rules = (
|
||||
# these two rules expects requests from start url
|
||||
Rule(r'imdb.com/nowplaying/$', 'parse_now_playing'),
|
||||
Rule(r'imdb.com/chart/top$', 'parse_top_250'),
|
||||
# this rule will parse requests manually dispatched
|
||||
Rule(r'imdb.com/title/tt\d+/$', 'parse_movie_info'),
|
||||
)
|
||||
|
||||
def parse_now_playing(self, response):
|
||||
"""Scrapes USA openings this week and top 10 in week"""
|
||||
self.log("Parsing USA Top Week")
|
||||
hxs = HtmlXPathSelector(response)
|
||||
|
||||
_urljoin = lambda url: self._urljoin(response, url)
|
||||
|
||||
#
|
||||
# openings this week
|
||||
#
|
||||
openings = hxs.select('//table[@class="movies"]//a[@class="title"]')
|
||||
boxoffice = hxs.select('//table[@class="boxoffice movies"]//a[@class="title"]')
|
||||
|
||||
opening_titles = openings.select('text()').extract()
|
||||
opening_urls = imap(_urljoin, openings.select('@href').extract())
|
||||
|
||||
box_titles = boxoffice.select('text()').extract()
|
||||
box_urls = imap(_urljoin, boxoffice.select('@href').extract())
|
||||
|
||||
# items
|
||||
opening_items = (UsaOpeningWeekMovie(title=title, url=url)
|
||||
for (title, url)
|
||||
in izip(opening_titles, opening_urls))
|
||||
|
||||
box_items = (UsaTopWeekMovie(title=title, url=url)
|
||||
for (title, url)
|
||||
in izip(box_titles, box_urls))
|
||||
|
||||
# movie requests
|
||||
requests = imap(self.make_requests_from_url,
|
||||
chain(opening_urls, box_urls))
|
||||
|
||||
return chain(opening_items, box_items, requests)
|
||||
|
||||
def parse_top_250(self, response):
|
||||
"""Scrapes movies from top 250 list"""
|
||||
self.log("Parsing Top 250")
|
||||
hxs = HtmlXPathSelector(response)
|
||||
|
||||
# scrap each row in the table
|
||||
rows = hxs.select('//div[@id="main"]/table/tr//a/ancestor::tr')
|
||||
for row in rows:
|
||||
fields = row.select('td//text()').extract()
|
||||
url, = row.select('td//a/@href').extract()
|
||||
url = self._urljoin(response, url)
|
||||
|
||||
item = Top250Movie()
|
||||
item['title'] = fields[2]
|
||||
item['url'] = url
|
||||
item['rank'] = fields[0]
|
||||
item['rating'] = fields[1]
|
||||
item['year'] = fields[3]
|
||||
item['votes'] = fields[4]
|
||||
|
||||
# scrapped top250 item
|
||||
yield item
|
||||
# fetch movie
|
||||
yield self.make_requests_from_url(url)
|
||||
|
||||
def parse_movie_info(self, response):
|
||||
"""Scrapes movie information"""
|
||||
self.log("Parsing Movie Info")
|
||||
hxs = HtmlXPathSelector(response)
|
||||
selector = hxs.select('//div[@class="maindetails"]')
|
||||
|
||||
item = MovieItem()
|
||||
# set url
|
||||
item['url'] = response.url
|
||||
|
||||
# use item loader for other attributes
|
||||
l = XPathItemLoader(item=item, selector=selector)
|
||||
l.add_xpath('title', './/h1/text()')
|
||||
l.add_xpath('release_date', './/h5[text()="Release Date:"]'
|
||||
'/following-sibling::div/text()')
|
||||
l.add_xpath('tagline', './/h5[text()="Tagline:"]'
|
||||
'/following-sibling::div/text()')
|
||||
|
||||
yield l.load_item()
|
||||
|
||||
def _urljoin(self, response, url):
|
||||
"""Helper to convert relative urls to absolute"""
|
||||
return urljoin_rfc(response.url, url, response.encoding)
|
||||
|
||||
SPIDER = ImdbSiteSpider()
|
||||
|
|
@ -0,0 +1,7 @@
|
|||
#!/usr/bin/env python
|
||||
|
||||
import os
|
||||
os.environ.setdefault('SCRAPY_SETTINGS_MODULE', 'imdb.settings')
|
||||
|
||||
from scrapy.command.cmdline import execute
|
||||
execute()
|
||||
|
|
@ -1,51 +0,0 @@
|
|||
"""
|
||||
Simple script to follow links from a start url. The links are followed in no
|
||||
particular order.
|
||||
|
||||
Usage:
|
||||
count_and_follow_links.py <start_url> <links_to_follow>
|
||||
|
||||
Example:
|
||||
count_and_follow_links.py http://scrapy.org/ 20
|
||||
|
||||
For each page visisted, this script will print the page body size and the
|
||||
number of links found.
|
||||
"""
|
||||
|
||||
import sys
|
||||
from urlparse import urljoin
|
||||
|
||||
from scrapy.crawler import Crawler
|
||||
from scrapy.selector import HtmlXPathSelector
|
||||
from scrapy.http import Request, HtmlResponse
|
||||
|
||||
links_followed = 0
|
||||
|
||||
def parse(response):
|
||||
global links_followed
|
||||
links_followed += 1
|
||||
if links_followed >= links_to_follow:
|
||||
crawler.stop()
|
||||
|
||||
# ignore non-HTML responses
|
||||
if not isinstance(response, HtmlResponse):
|
||||
return
|
||||
|
||||
links = HtmlXPathSelector(response).select('//a/@href').extract()
|
||||
abslinks = [urljoin(response.url, l) for l in links]
|
||||
|
||||
print "page %2d/%d: %s" % (links_followed, links_to_follow, response.url)
|
||||
print " size : %d bytes" % len(response.body)
|
||||
print " links: %d" % len(links)
|
||||
print
|
||||
|
||||
return [Request(l, callback=parse) for l in abslinks]
|
||||
|
||||
if len(sys.argv) != 3:
|
||||
print __doc__
|
||||
sys.exit(2)
|
||||
|
||||
start_url, links_to_follow = sys.argv[1], int(sys.argv[2])
|
||||
request = Request(start_url, callback=parse)
|
||||
crawler = Crawler()
|
||||
crawler.crawl(request)
|
||||
|
|
@ -2,8 +2,8 @@
|
|||
Scrapy - a screen scraping framework written in Python
|
||||
"""
|
||||
|
||||
version_info = (0, 8, 0, '', 0)
|
||||
__version__ = "0.8"
|
||||
version_info = (0, 9, 0, 'dev')
|
||||
__version__ = "0.9-dev"
|
||||
|
||||
import sys, os, warnings
|
||||
|
||||
|
|
@ -17,11 +17,6 @@ warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
|
|||
# monkey patches to fix external library issues
|
||||
from scrapy.xlib import twisted_250_monkeypatches
|
||||
|
||||
# add some common encoding aliases not included by default in Python
|
||||
from scrapy.utils.encoding import add_encoding_alias
|
||||
add_encoding_alias('gb2312', 'zh-cn')
|
||||
add_encoding_alias('cp1251', 'win-1251')
|
||||
|
||||
# optional_features is a set containing Scrapy optional features
|
||||
optional_features = set()
|
||||
|
||||
|
|
|
|||
|
|
@ -7,20 +7,14 @@ import cProfile
|
|||
|
||||
import scrapy
|
||||
from scrapy import log
|
||||
from scrapy.spider import spiders
|
||||
from scrapy.xlib import lsprofcalltree
|
||||
from scrapy.conf import settings
|
||||
from scrapy.command.models import ScrapyCommand
|
||||
from scrapy.utils.signal import send_catch_log
|
||||
|
||||
# This dict holds information about the executed command for later use
|
||||
command_executed = {}
|
||||
|
||||
def _save_command_executed(cmdname, cmd, args, opts):
|
||||
"""Save command executed info for later reference"""
|
||||
command_executed['name'] = cmdname
|
||||
command_executed['class'] = cmd
|
||||
command_executed['args'] = args[:]
|
||||
command_executed['opts'] = opts.__dict__.copy()
|
||||
# Signal that carries information about the command which was executed
|
||||
# args: cmdname, cmdobj, args, opts
|
||||
command_executed = object()
|
||||
|
||||
def _find_commands(dir):
|
||||
try:
|
||||
|
|
@ -127,7 +121,8 @@ def execute(argv=None):
|
|||
sys.exit(2)
|
||||
|
||||
del args[0] # remove command name from args
|
||||
_save_command_executed(cmdname, cmd, args, opts)
|
||||
send_catch_log(signal=command_executed, cmdname=cmdname, cmdobj=cmd, \
|
||||
args=args, opts=opts)
|
||||
from scrapy.core.manager import scrapymanager
|
||||
scrapymanager.configure(control_reactor=True)
|
||||
ret = _run_command(cmd, args, opts)
|
||||
|
|
@ -136,23 +131,25 @@ def execute(argv=None):
|
|||
|
||||
def _run_command(cmd, args, opts):
|
||||
if opts.profile or opts.lsprof:
|
||||
if opts.profile:
|
||||
log.msg("writing cProfile stats to %r" % opts.profile)
|
||||
if opts.lsprof:
|
||||
log.msg("writing lsprof stats to %r" % opts.lsprof)
|
||||
loc = locals()
|
||||
p = cProfile.Profile()
|
||||
p.runctx('ret = cmd.run(args, opts)', globals(), loc)
|
||||
if opts.profile:
|
||||
p.dump_stats(opts.profile)
|
||||
k = lsprofcalltree.KCacheGrind(p)
|
||||
if opts.lsprof:
|
||||
with open(opts.lsprof, 'w') as f:
|
||||
k.output(f)
|
||||
ret = loc['ret']
|
||||
return _run_command_profiled(cmd, args, opts)
|
||||
else:
|
||||
ret = cmd.run(args, opts)
|
||||
return ret
|
||||
return cmd.run(args, opts)
|
||||
|
||||
def _run_command_profiled(cmd, args, opts):
|
||||
if opts.profile:
|
||||
log.msg("writing cProfile stats to %r" % opts.profile)
|
||||
if opts.lsprof:
|
||||
log.msg("writing lsprof stats to %r" % opts.lsprof)
|
||||
loc = locals()
|
||||
p = cProfile.Profile()
|
||||
p.runctx('ret = cmd.run(args, opts)', globals(), loc)
|
||||
if opts.profile:
|
||||
p.dump_stats(opts.profile)
|
||||
k = lsprofcalltree.KCacheGrind(p)
|
||||
if opts.lsprof:
|
||||
with open(opts.lsprof, 'w') as f:
|
||||
k.output(f)
|
||||
return loc['ret']
|
||||
|
||||
if __name__ == '__main__':
|
||||
execute()
|
||||
|
|
|
|||
|
|
@ -71,6 +71,23 @@ DOWNLOADER_STATS = True
|
|||
|
||||
DUPEFILTER_CLASS = 'scrapy.contrib.dupefilter.RequestFingerprintDupeFilter'
|
||||
|
||||
ENCODING_ALIASES = {}
|
||||
|
||||
ENCODING_ALIASES_BASE = {
|
||||
'zh-cn': 'gb2312',
|
||||
'win-1251': 'cp1251',
|
||||
'macintosh' : 'mac-roman',
|
||||
'x-sjis': 'shift-jis',
|
||||
'iso-8859-1': 'cp1252',
|
||||
'iso8859-1': 'cp1252',
|
||||
'8859': 'cp1252',
|
||||
'cp819': 'cp1252',
|
||||
'latin': 'cp1252',
|
||||
'latin1': 'cp1252',
|
||||
'latin_1': 'cp1252',
|
||||
'l1': 'cp1252',
|
||||
}
|
||||
|
||||
EXTENSIONS = {}
|
||||
|
||||
EXTENSIONS_BASE = {
|
||||
|
|
@ -101,6 +118,7 @@ ITEM_PROCESSOR = 'scrapy.contrib.pipeline.ItemPipelineManager'
|
|||
ITEM_PIPELINES = []
|
||||
|
||||
LOG_ENABLED = True
|
||||
LOG_ENCODING = 'utf-8'
|
||||
LOG_FORMATTER_CRAWLED = 'scrapy.contrib.logformatter.crawled_logline'
|
||||
LOG_STDOUT = False
|
||||
LOG_LEVEL = 'DEBUG'
|
||||
|
|
@ -122,6 +140,8 @@ MYSQL_CONNECTION_SETTINGS = {}
|
|||
|
||||
NEWSPIDER_MODULE = ''
|
||||
|
||||
RANDOMIZE_DOWNLOAD_DELAY = True
|
||||
|
||||
REDIRECT_MAX_METAREFRESH_DELAY = 100
|
||||
REDIRECT_MAX_TIMES = 20 # uses Firefox default setting
|
||||
REDIRECT_PRIORITY_ADJUST = +2
|
||||
|
|
@ -150,7 +170,7 @@ SCHEDULER_MIDDLEWARES_BASE = {
|
|||
'scrapy.contrib.schedulermiddleware.duplicatesfilter.DuplicatesFilterMiddleware': 500,
|
||||
}
|
||||
|
||||
SCHEDULER_ORDER = 'BFO' # available orders: BFO (default), DFO
|
||||
SCHEDULER_ORDER = 'DFO'
|
||||
|
||||
SPIDER_MANAGER_CLASS = 'scrapy.contrib.spidermanager.TwistedPluginSpiderManager'
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,5 @@
|
|||
from scrapy import log
|
||||
from scrapy.http import HtmlResponse
|
||||
from scrapy.utils.url import urljoin_rfc
|
||||
from scrapy.utils.response import get_meta_refresh
|
||||
from scrapy.core.exceptions import IgnoreRequest
|
||||
|
|
@ -24,10 +25,11 @@ class RedirectMiddleware(object):
|
|||
redirected = request.replace(url=redirected_url)
|
||||
return self._redirect(redirected, request, spider, response.status)
|
||||
|
||||
interval, url = get_meta_refresh(response)
|
||||
if url and interval < self.max_metarefresh_delay:
|
||||
redirected = self._redirect_request_using_get(request, url)
|
||||
return self._redirect(redirected, request, spider, 'meta refresh')
|
||||
if isinstance(response, HtmlResponse):
|
||||
interval, url = get_meta_refresh(response)
|
||||
if url and interval < self.max_metarefresh_delay:
|
||||
redirected = self._redirect_request_using_get(request, url)
|
||||
return self._redirect(redirected, request, spider, 'meta refresh')
|
||||
|
||||
return response
|
||||
|
||||
|
|
|
|||
|
|
@ -1,26 +0,0 @@
|
|||
"""
|
||||
Extensions to override scrapy settings with per-group settings according to the
|
||||
group the spider belongs to. It only overrides the settings when running the
|
||||
crawl command with *only one domain as argument*.
|
||||
"""
|
||||
|
||||
from scrapy.conf import settings
|
||||
from scrapy.core.exceptions import NotConfigured
|
||||
from scrapy.command.cmdline import command_executed
|
||||
|
||||
class GroupSettings(object):
|
||||
|
||||
def __init__(self):
|
||||
if not settings.getbool("GROUPSETTINGS_ENABLED"):
|
||||
raise NotConfigured
|
||||
|
||||
if command_executed and command_executed['name'] == 'crawl':
|
||||
mod = __import__(settings['GROUPSETTINGS_MODULE'], {}, {}, [''])
|
||||
args = command_executed['args']
|
||||
if len(args) == 1 and not args[0].startswith('http://'):
|
||||
domain = args[0]
|
||||
settings.overrides.update(mod.default_settings)
|
||||
for group, domains in mod.group_spiders.iteritems():
|
||||
if domain in domains:
|
||||
settings.overrides.update(mod.group_settings.get(group, {}))
|
||||
|
||||
|
|
@ -26,7 +26,7 @@ class HtmlParserLinkExtractor(HTMLParser):
|
|||
links = unique_list(self.links, key=lambda link: link.url) if self.unique else self.links
|
||||
|
||||
ret = []
|
||||
base_url = self.base_url if self.base_url else response_url
|
||||
base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
|
||||
for link in links:
|
||||
link.url = urljoin_rfc(base_url, link.url, response_encoding)
|
||||
link.url = safe_url_string(link.url, response_encoding)
|
||||
|
|
|
|||
|
|
@ -3,7 +3,6 @@ This module implements the HtmlImageLinkExtractor for extracting
|
|||
image links only.
|
||||
"""
|
||||
|
||||
import urlparse
|
||||
|
||||
from scrapy.link import Link
|
||||
from scrapy.utils.url import canonicalize_url, urljoin_rfc
|
||||
|
|
@ -25,13 +24,13 @@ class HTMLImageLinkExtractor(object):
|
|||
self.unique = unique
|
||||
self.canonicalize = canonicalize
|
||||
|
||||
def extract_from_selector(self, selector, parent=None):
|
||||
def extract_from_selector(self, selector, encoding, parent=None):
|
||||
ret = []
|
||||
def _add_link(url_sel, alt_sel=None):
|
||||
url = flatten([url_sel.extract()])
|
||||
alt = flatten([alt_sel.extract()]) if alt_sel else (u'', )
|
||||
if url:
|
||||
ret.append(Link(unicode_to_str(url[0]), alt[0]))
|
||||
ret.append(Link(unicode_to_str(url[0], encoding), alt[0]))
|
||||
|
||||
if selector.xmlNode.type == 'element':
|
||||
if selector.xmlNode.name == 'img':
|
||||
|
|
@ -41,7 +40,7 @@ class HTMLImageLinkExtractor(object):
|
|||
children = selector.select('child::*')
|
||||
if len(children):
|
||||
for child in children:
|
||||
ret.extend(self.extract_from_selector(child, parent=selector))
|
||||
ret.extend(self.extract_from_selector(child, encoding, parent=selector))
|
||||
elif selector.xmlNode.name == 'a' and not parent:
|
||||
_add_link(selector.select('@href'), selector.select('@title'))
|
||||
else:
|
||||
|
|
@ -52,7 +51,7 @@ class HTMLImageLinkExtractor(object):
|
|||
def extract_links(self, response):
|
||||
xs = HtmlXPathSelector(response)
|
||||
base_url = xs.select('//base/@href').extract()
|
||||
base_url = unicode_to_str(base_url[0]) if base_url else unicode_to_str(response.url)
|
||||
base_url = urljoin_rfc(response.url, base_url[0]) if base_url else response.url
|
||||
|
||||
links = []
|
||||
for location in self.locations:
|
||||
|
|
@ -64,7 +63,7 @@ class HTMLImageLinkExtractor(object):
|
|||
continue
|
||||
|
||||
for selector in selectors:
|
||||
links.extend(self.extract_from_selector(selector))
|
||||
links.extend(self.extract_from_selector(selector, response.encoding))
|
||||
|
||||
seen, ret = set(), []
|
||||
for link in links:
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ class LxmlLinkExtractor(object):
|
|||
links = unique_list(self.links, key=lambda link: link.url) if self.unique else self.links
|
||||
|
||||
ret = []
|
||||
base_url = self.base_url if self.base_url else response_url
|
||||
base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
|
||||
for link in links:
|
||||
link.url = urljoin_rfc(base_url, link.url, response_encoding)
|
||||
link.url = safe_url_string(link.url, response_encoding)
|
||||
|
|
|
|||
|
|
@ -16,8 +16,9 @@ def clean_link(link_text):
|
|||
|
||||
class RegexLinkExtractor(SgmlLinkExtractor):
|
||||
"""High performant link extractor"""
|
||||
|
||||
def _extract_links(self, response_text, response_url, response_encoding):
|
||||
base_url = self.base_url if self.base_url else response_url
|
||||
base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
|
||||
|
||||
clean_url = lambda u: urljoin_rfc(base_url, remove_entities(clean_link(u.decode(response_encoding))))
|
||||
clean_text = lambda t: replace_escape_chars(remove_tags(t.decode(response_encoding))).strip()
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ class BaseSgmlLinkExtractor(FixedSGMLParser):
|
|||
links = unique_list(self.links, key=lambda link: link.url) if self.unique else self.links
|
||||
|
||||
ret = []
|
||||
base_url = self.base_url if self.base_url else response_url
|
||||
base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
|
||||
for link in links:
|
||||
link.url = urljoin_rfc(base_url, link.url, response_encoding)
|
||||
link.url = safe_url_string(link.url, response_encoding)
|
||||
|
|
|
|||
|
|
@ -0,0 +1,4 @@
|
|||
"""CrawlSpider v2"""
|
||||
|
||||
from .rules import Rule
|
||||
from .spider import CrawlSpider
|
||||
|
|
@ -0,0 +1,61 @@
|
|||
"""
|
||||
Request/Response Matchers
|
||||
|
||||
Perform evaluation to Request or Response attributes
|
||||
"""
|
||||
|
||||
import re
|
||||
|
||||
class BaseMatcher(object):
|
||||
"""Base matcher. Returns True by default."""
|
||||
|
||||
def matches_request(self, request):
|
||||
"""Performs Request Matching"""
|
||||
return True
|
||||
|
||||
def matches_response(self, response):
|
||||
"""Performs Response Matching"""
|
||||
return True
|
||||
|
||||
|
||||
class UrlMatcher(BaseMatcher):
|
||||
"""Matches URL attribute"""
|
||||
|
||||
def __init__(self, url):
|
||||
"""Initialize url attribute"""
|
||||
self._url = url
|
||||
|
||||
def matches_url(self, url):
|
||||
"""Returns True if given url is equal to matcher's url"""
|
||||
return self._url == url
|
||||
|
||||
def matches_request(self, request):
|
||||
"""Returns True if Request's url matches initial url"""
|
||||
return self.matches_url(request.url)
|
||||
|
||||
def matches_response(self, response):
|
||||
"""Returns True if Response's url matches initial url"""
|
||||
return self.matches_url(response.url)
|
||||
|
||||
|
||||
class UrlRegexMatcher(UrlMatcher):
|
||||
"""Matches URL using regular expression"""
|
||||
|
||||
def __init__(self, regex, flags=0):
|
||||
"""Initialize regular expression"""
|
||||
self._regex = re.compile(regex, flags)
|
||||
|
||||
def matches_url(self, url):
|
||||
"""Returns True if url matches regular expression"""
|
||||
return self._regex.search(url) is not None
|
||||
|
||||
|
||||
class UrlListMatcher(UrlMatcher):
|
||||
"""Matches if URL is in List"""
|
||||
|
||||
def __init__(self, urls):
|
||||
self._urls = urls
|
||||
|
||||
def matches_url(self, url):
|
||||
"""Returns True if url is in urls list"""
|
||||
return url in self._urls
|
||||
|
|
@ -0,0 +1,117 @@
|
|||
"""Request Extractors"""
|
||||
from scrapy.http import Request
|
||||
from scrapy.selector import HtmlXPathSelector
|
||||
from scrapy.utils.misc import arg_to_iter
|
||||
from scrapy.utils.python import FixedSGMLParser, str_to_unicode
|
||||
from scrapy.utils.url import safe_url_string, urljoin_rfc
|
||||
|
||||
from itertools import ifilter
|
||||
|
||||
|
||||
class BaseSgmlRequestExtractor(FixedSGMLParser):
|
||||
"""Base SGML Request Extractor"""
|
||||
|
||||
def __init__(self, tag='a', attr='href'):
|
||||
"""Initialize attributes"""
|
||||
FixedSGMLParser.__init__(self)
|
||||
|
||||
self.scan_tag = tag if callable(tag) else lambda t: t == tag
|
||||
self.scan_attr = attr if callable(attr) else lambda a: a == attr
|
||||
self.current_request = None
|
||||
|
||||
def extract_requests(self, response):
|
||||
"""Returns list of requests extracted from response"""
|
||||
return self._extract_requests(response.body, response.url,
|
||||
response.encoding)
|
||||
|
||||
def _extract_requests(self, response_text, response_url, response_encoding):
|
||||
"""Extract requests with absolute urls"""
|
||||
self.reset()
|
||||
self.feed(response_text)
|
||||
self.close()
|
||||
|
||||
base_url = urljoin_rfc(response_url, self.base_url) if self.base_url else response_url
|
||||
self._make_absolute_urls(base_url, response_encoding)
|
||||
self._fix_link_text_encoding(response_encoding)
|
||||
|
||||
return self.requests
|
||||
|
||||
def _make_absolute_urls(self, base_url, encoding):
|
||||
"""Makes all request's urls absolute"""
|
||||
for req in self.requests:
|
||||
url = req.url
|
||||
# make absolute url
|
||||
url = urljoin_rfc(base_url, url, encoding)
|
||||
url = safe_url_string(url, encoding)
|
||||
# replace in-place request's url
|
||||
req.url = url
|
||||
|
||||
def _fix_link_text_encoding(self, encoding):
|
||||
"""Convert link_text to unicode for each request"""
|
||||
for req in self.requests:
|
||||
req.meta.setdefault('link_text', '')
|
||||
req.meta['link_text'] = str_to_unicode(req.meta['link_text'],
|
||||
encoding)
|
||||
|
||||
def reset(self):
|
||||
"""Reset state"""
|
||||
FixedSGMLParser.reset(self)
|
||||
self.requests = []
|
||||
self.base_url = None
|
||||
|
||||
def unknown_starttag(self, tag, attrs):
|
||||
"""Process unknown start tag"""
|
||||
if 'base' == tag:
|
||||
self.base_url = dict(attrs).get('href')
|
||||
|
||||
_matches = lambda (attr, value): self.scan_attr(attr) \
|
||||
and value is not None
|
||||
if self.scan_tag(tag):
|
||||
for attr, value in ifilter(_matches, attrs):
|
||||
req = Request(url=value)
|
||||
self.requests.append(req)
|
||||
self.current_request = req
|
||||
|
||||
def unknown_endtag(self, tag):
|
||||
"""Process unknown end tag"""
|
||||
self.current_request = None
|
||||
|
||||
def handle_data(self, data):
|
||||
"""Process data"""
|
||||
current = self.current_request
|
||||
if current and not 'link_text' in current.meta:
|
||||
current.meta['link_text'] = data.strip()
|
||||
|
||||
|
||||
class SgmlRequestExtractor(BaseSgmlRequestExtractor):
|
||||
"""SGML Request Extractor"""
|
||||
|
||||
def __init__(self, tags=None, attrs=None):
|
||||
"""Initialize with custom tag & attribute function checkers"""
|
||||
# defaults
|
||||
tags = tuple(tags) if tags else ('a', 'area')
|
||||
attrs = tuple(attrs) if attrs else ('href', )
|
||||
|
||||
tag_func = lambda x: x in tags
|
||||
attr_func = lambda x: x in attrs
|
||||
BaseSgmlRequestExtractor.__init__(self, tag=tag_func, attr=attr_func)
|
||||
|
||||
# TODO: move to own file
|
||||
class XPathRequestExtractor(SgmlRequestExtractor):
|
||||
"""SGML Request Extractor with XPath restriction"""
|
||||
|
||||
def __init__(self, restrict_xpaths, tags=None, attrs=None):
|
||||
"""Initialize XPath restrictions"""
|
||||
self.restrict_xpaths = tuple(arg_to_iter(restrict_xpaths))
|
||||
SgmlRequestExtractor.__init__(self, tags, attrs)
|
||||
|
||||
def extract_requests(self, response):
|
||||
"""Restrict to XPath regions"""
|
||||
hxs = HtmlXPathSelector(response)
|
||||
fragments = (''.join(
|
||||
html_frag for html_frag in hxs.select(xpath).extract()
|
||||
) for xpath in self.restrict_xpaths)
|
||||
html_slice = ''.join(html_frag for html_frag in fragments)
|
||||
return self._extract_requests(html_slice, response.url,
|
||||
response.encoding)
|
||||
|
||||
|
|
@ -0,0 +1,27 @@
|
|||
"""Request Generator"""
|
||||
from itertools import imap
|
||||
|
||||
class RequestGenerator(object):
|
||||
"""Extracto and process requests from response"""
|
||||
|
||||
def __init__(self, req_extractors, req_processors, callback, spider=None):
|
||||
"""Initialize attributes"""
|
||||
self._request_extractors = req_extractors
|
||||
self._request_processors = req_processors
|
||||
#TODO: resolve callback?
|
||||
self._callback = callback
|
||||
|
||||
def generate_requests(self, response):
|
||||
"""Extract and process new requests from response.
|
||||
Attach callback to each request as default callback."""
|
||||
requests = []
|
||||
for ext in self._request_extractors:
|
||||
requests.extend(ext.extract_requests(response))
|
||||
|
||||
for proc in self._request_processors:
|
||||
requests = proc(requests)
|
||||
|
||||
# return iterator
|
||||
# @@@ creates new Request object with callback
|
||||
return imap(lambda r: r.replace(callback=self._callback), requests)
|
||||
|
||||
|
|
@ -0,0 +1,111 @@
|
|||
"""Request Processors"""
|
||||
from scrapy.utils.misc import arg_to_iter
|
||||
from scrapy.utils.url import canonicalize_url, url_is_from_any_domain
|
||||
|
||||
from itertools import ifilter, imap
|
||||
|
||||
import re
|
||||
|
||||
class Canonicalize(object):
|
||||
"""Canonicalize Request Processor"""
|
||||
def _replace_url(self, req):
|
||||
# replace in-place
|
||||
req.url = canonicalize_url(req.url)
|
||||
return req
|
||||
|
||||
def __call__(self, requests):
|
||||
"""Canonicalize all requests' urls"""
|
||||
return imap(self._replace_url, requests)
|
||||
|
||||
|
||||
class FilterDupes(object):
|
||||
"""Filter duplicate Requests"""
|
||||
|
||||
def __init__(self, *attributes):
|
||||
"""Initialize comparison attributes"""
|
||||
self._attributes = tuple(attributes) if attributes \
|
||||
else tuple(['url'])
|
||||
|
||||
def _equal_attr(self, obj1, obj2, attr):
|
||||
return getattr(obj1, attr) == getattr(obj2, attr)
|
||||
|
||||
def _requests_equal(self, req1, req2):
|
||||
"""Attribute comparison helper"""
|
||||
# look for not equal attribute
|
||||
_not_equal = lambda attr: not self._equal_attr(req1, req2, attr)
|
||||
for attr in ifilter(_not_equal, self._attributes):
|
||||
return False
|
||||
# all attributes equal
|
||||
return True
|
||||
|
||||
def _request_in(self, request, requests_seen):
|
||||
"""Check if request is in given requests seen list"""
|
||||
_req_seen = lambda r: self._requests_equal(r, request)
|
||||
for seen in ifilter(_req_seen, requests_seen):
|
||||
return True
|
||||
# request not seen
|
||||
return False
|
||||
|
||||
def __call__(self, requests):
|
||||
"""Filter seen requests"""
|
||||
# per-call duplicates filter
|
||||
self.requests_seen = set()
|
||||
_not_seen = lambda r: not self._request_in(r, self.requests_seen)
|
||||
for req in ifilter(_not_seen, requests):
|
||||
yield req
|
||||
# registry seen request
|
||||
self.requests_seen.add(req)
|
||||
|
||||
|
||||
class FilterDomain(object):
|
||||
"""Filter request's domain"""
|
||||
|
||||
def __init__(self, allow=(), deny=()):
|
||||
"""Initialize allow/deny attributes"""
|
||||
self.allow = tuple(arg_to_iter(allow))
|
||||
self.deny = tuple(arg_to_iter(deny))
|
||||
|
||||
def __call__(self, requests):
|
||||
"""Filter domains"""
|
||||
processed = (req for req in requests)
|
||||
|
||||
if self.allow:
|
||||
processed = (req for req in requests
|
||||
if url_is_from_any_domain(req.url, self.allow))
|
||||
if self.deny:
|
||||
processed = (req for req in requests
|
||||
if not url_is_from_any_domain(req.url, self.deny))
|
||||
|
||||
return processed
|
||||
|
||||
|
||||
class FilterUrl(object):
|
||||
"""Filter request's url"""
|
||||
|
||||
def __init__(self, allow=(), deny=()):
|
||||
"""Initialize allow/deny attributes"""
|
||||
_re_type = type(re.compile('', 0))
|
||||
|
||||
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(allow)]
|
||||
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
|
||||
for x in arg_to_iter(deny)]
|
||||
|
||||
def __call__(self, requests):
|
||||
"""Filter request's url based on allow/deny rules"""
|
||||
#TODO: filter valid urls here?
|
||||
processed = (req for req in requests)
|
||||
|
||||
if self.allow_res:
|
||||
processed = (req for req in requests
|
||||
if self._matches(req.url, self.allow_res))
|
||||
if self.deny_res:
|
||||
processed = (req for req in requests
|
||||
if not self._matches(req.url, self.deny_res))
|
||||
|
||||
return processed
|
||||
|
||||
def _matches(self, url, regexs):
|
||||
"""Returns True if url matches any regex in given list"""
|
||||
return any(r.search(url) for r in regexs)
|
||||
|
||||
|
|
@ -0,0 +1,100 @@
|
|||
"""Crawler Rules"""
|
||||
from scrapy.http import Request
|
||||
from scrapy.http import Response
|
||||
|
||||
from functools import partial
|
||||
from itertools import ifilter
|
||||
|
||||
from .matchers import BaseMatcher
|
||||
# default strint-to-matcher class
|
||||
from .matchers import UrlRegexMatcher
|
||||
|
||||
class CompiledRule(object):
|
||||
"""Compiled version of Rule"""
|
||||
def __init__(self, matcher, callback=None, follow=False):
|
||||
"""Initialize attributes checking type"""
|
||||
assert isinstance(matcher, BaseMatcher)
|
||||
assert callback is None or callable(callback)
|
||||
assert isinstance(follow, bool)
|
||||
|
||||
self.matcher = matcher
|
||||
self.callback = callback
|
||||
self.follow = follow
|
||||
|
||||
|
||||
class Rule(object):
|
||||
"""Crawler Rule"""
|
||||
def __init__(self, matcher=None, callback=None, follow=False, **kwargs):
|
||||
"""Store attributes"""
|
||||
self.matcher = matcher
|
||||
self.callback = callback
|
||||
self.cb_kwargs = kwargs if kwargs else {}
|
||||
self.follow = True if follow else False
|
||||
|
||||
if self.callback is None and self.follow is False:
|
||||
raise ValueError("Rule must either have a callback or "
|
||||
"follow=True: %r" % self)
|
||||
|
||||
def __repr__(self):
|
||||
return "Rule(matcher=%r, callback=%r, follow=%r, **%r)" \
|
||||
% (self.matcher, self.callback, self.follow, self.cb_kwargs)
|
||||
|
||||
|
||||
class RulesManager(object):
|
||||
"""Rules Manager"""
|
||||
def __init__(self, rules, spider, default_matcher=UrlRegexMatcher):
|
||||
"""Initialize rules using spider and default matcher"""
|
||||
self._rules = tuple()
|
||||
|
||||
# compile absolute/relative-to-spider callbacks"""
|
||||
for rule in rules:
|
||||
# prepare matcher
|
||||
if rule.matcher is None:
|
||||
# instance BaseMatcher by default
|
||||
matcher = BaseMatcher()
|
||||
elif isinstance(rule.matcher, BaseMatcher):
|
||||
matcher = rule.matcher
|
||||
else:
|
||||
# matcher not BaseMatcher, check for string
|
||||
if isinstance(rule.matcher, basestring):
|
||||
# instance default matcher
|
||||
matcher = default_matcher(rule.matcher)
|
||||
else:
|
||||
raise ValueError('Not valid matcher given %r in %r' \
|
||||
% (rule.matcher, rule))
|
||||
|
||||
# prepare callback
|
||||
if callable(rule.callback):
|
||||
callback = rule.callback
|
||||
elif not rule.callback is None:
|
||||
# callback from spider
|
||||
callback = getattr(spider, rule.callback)
|
||||
|
||||
if not callable(callback):
|
||||
raise AttributeError('Invalid callback %r can not be resolved' \
|
||||
% callback)
|
||||
else:
|
||||
callback = None
|
||||
|
||||
if rule.cb_kwargs:
|
||||
# build partial callback
|
||||
callback = partial(callback, **rule.cb_kwargs)
|
||||
|
||||
# append compiled rule to rules list
|
||||
crule = CompiledRule(matcher, callback, follow=rule.follow)
|
||||
self._rules += (crule, )
|
||||
|
||||
def get_rule_from_request(self, request):
|
||||
"""Returns first rule that matches given Request"""
|
||||
_matches = lambda r: r.matcher.matches_request(request)
|
||||
for rule in ifilter(_matches, self._rules):
|
||||
# return first match of iterator
|
||||
return rule
|
||||
|
||||
def get_rule_from_response(self, response):
|
||||
"""Returns first rule that matches given Response"""
|
||||
_matches = lambda r: r.matcher.matches_response(response)
|
||||
for rule in ifilter(_matches, self._rules):
|
||||
# return first match of iterator
|
||||
return rule
|
||||
|
||||
|
|
@ -0,0 +1,69 @@
|
|||
"""CrawlSpider v2"""
|
||||
from scrapy.spider import BaseSpider
|
||||
from scrapy.utils.spider import iterate_spider_output
|
||||
|
||||
from .matchers import UrlListMatcher
|
||||
from .rules import Rule, RulesManager
|
||||
from .reqext import SgmlRequestExtractor
|
||||
from .reqgen import RequestGenerator
|
||||
from .reqproc import Canonicalize, FilterDupes
|
||||
|
||||
class CrawlSpider(BaseSpider):
|
||||
"""CrawlSpider v2"""
|
||||
|
||||
request_extractors = None
|
||||
request_processors = None
|
||||
rules = []
|
||||
|
||||
def __init__(self):
|
||||
"""Initialize dispatcher"""
|
||||
super(CrawlSpider, self).__init__()
|
||||
|
||||
# auto follow start urls
|
||||
if self.start_urls:
|
||||
_matcher = UrlListMatcher(self.start_urls)
|
||||
# append new rule using type from current self.rules
|
||||
rules = self.rules + type(self.rules)([
|
||||
Rule(_matcher, follow=True)
|
||||
])
|
||||
else:
|
||||
rules = self.rules
|
||||
|
||||
# set defaults if not set
|
||||
if self.request_extractors is None:
|
||||
# default link extractor. Extracts all links from response
|
||||
self.request_extractors = [ SgmlRequestExtractor() ]
|
||||
|
||||
if self.request_processors is None:
|
||||
# default proccessor. Filter duplicates requests
|
||||
self.request_processors = [ FilterDupes() ]
|
||||
|
||||
|
||||
# wrap rules
|
||||
self._rulesman = RulesManager(rules, spider=self)
|
||||
# generates new requests with given callback
|
||||
self._reqgen = RequestGenerator(self.request_extractors,
|
||||
self.request_processors,
|
||||
callback=self.parse)
|
||||
|
||||
def parse(self, response):
|
||||
"""Dispatch callback and generate requests"""
|
||||
# get rule for response
|
||||
rule = self._rulesman.get_rule_from_response(response)
|
||||
|
||||
if rule:
|
||||
# dispatch callback if set
|
||||
if rule.callback:
|
||||
output = iterate_spider_output(rule.callback(response))
|
||||
for req_or_item in output:
|
||||
yield req_or_item
|
||||
|
||||
if rule.follow:
|
||||
for req in self._reqgen.generate_requests(response):
|
||||
# only dispatch request if has matching rule
|
||||
if self._rulesman.get_rule_from_request(req):
|
||||
yield req
|
||||
else:
|
||||
self.log("No rule for response %s" % response, level=log.WARNING)
|
||||
|
||||
|
||||
|
|
@ -1,55 +0,0 @@
|
|||
"""
|
||||
A pipeline to persist objects using shove.
|
||||
|
||||
Shove is a "new generation" shelve. For more information see:
|
||||
http://pypi.python.org/pypi/shove
|
||||
"""
|
||||
|
||||
from string import Template
|
||||
|
||||
from shove import Shove
|
||||
from scrapy.xlib.pydispatch import dispatcher
|
||||
|
||||
from scrapy import log
|
||||
from scrapy.core import signals
|
||||
from scrapy.conf import settings
|
||||
from scrapy.core.exceptions import NotConfigured
|
||||
|
||||
class ShoveItemPipeline(object):
|
||||
|
||||
def __init__(self):
|
||||
self.uritpl = settings['SHOVEITEM_STORE_URI']
|
||||
if not self.uritpl:
|
||||
raise NotConfigured
|
||||
self.opts = settings['SHOVEITEM_STORE_OPT'] or {}
|
||||
self.stores = {}
|
||||
|
||||
dispatcher.connect(self.spider_opened, signal=signals.spider_opened)
|
||||
dispatcher.connect(self.spider_closed, signal=signals.spider_closed)
|
||||
|
||||
def process_item(self, spider, item):
|
||||
guid = str(item.guid)
|
||||
|
||||
if guid in self.stores[spider]:
|
||||
if self.stores[spider][guid] == item:
|
||||
status = 'old'
|
||||
else:
|
||||
status = 'upd'
|
||||
else:
|
||||
status = 'new'
|
||||
|
||||
if not status == 'old':
|
||||
self.stores[spider][guid] = item
|
||||
self.log(spider, item, status)
|
||||
return item
|
||||
|
||||
def spider_opened(self, spider):
|
||||
uri = Template(self.uritpl).substitute(domain=spider.domain_name)
|
||||
self.stores[spider] = Shove(uri, **self.opts)
|
||||
|
||||
def spider_closed(self, spider):
|
||||
self.stores[spider].sync()
|
||||
|
||||
def log(self, spider, item, status):
|
||||
log.msg("Shove (%s): Item guid=%s" % (status, item.guid), level=log.DEBUG, \
|
||||
spider=spider)
|
||||
|
|
@ -2,6 +2,7 @@
|
|||
Download web pages using asynchronous IO
|
||||
"""
|
||||
|
||||
import random
|
||||
from time import time
|
||||
|
||||
from twisted.internet import reactor, defer
|
||||
|
|
@ -20,15 +21,21 @@ class SpiderInfo(object):
|
|||
|
||||
def __init__(self, download_delay=None, max_concurrent_requests=None):
|
||||
if download_delay is None:
|
||||
self.download_delay = settings.getfloat('DOWNLOAD_DELAY')
|
||||
self._download_delay = settings.getfloat('DOWNLOAD_DELAY')
|
||||
else:
|
||||
self.download_delay = download_delay
|
||||
if self.download_delay:
|
||||
self._download_delay = float(download_delay)
|
||||
if self._download_delay:
|
||||
self.max_concurrent_requests = 1
|
||||
elif max_concurrent_requests is None:
|
||||
self.max_concurrent_requests = settings.getint('CONCURRENT_REQUESTS_PER_SPIDER')
|
||||
else:
|
||||
self.max_concurrent_requests = max_concurrent_requests
|
||||
if self._download_delay and settings.getbool('RANDOMIZE_DOWNLOAD_DELAY'):
|
||||
# same policy as wget --random-wait
|
||||
self.random_delay_interval = (0.5*self._download_delay, \
|
||||
1.5*self._download_delay)
|
||||
else:
|
||||
self.random_delay_interval = None
|
||||
|
||||
self.active = set()
|
||||
self.queue = []
|
||||
|
|
@ -44,6 +51,12 @@ class SpiderInfo(object):
|
|||
# use self.active to include requests in the downloader middleware
|
||||
return len(self.active) > 2 * self.max_concurrent_requests
|
||||
|
||||
def download_delay(self):
|
||||
if self.random_delay_interval:
|
||||
return random.uniform(*self.random_delay_interval)
|
||||
else:
|
||||
return self._download_delay
|
||||
|
||||
def cancel_request_calls(self):
|
||||
for call in self.next_request_calls:
|
||||
call.cancel()
|
||||
|
|
@ -99,8 +112,9 @@ class Downloader(object):
|
|||
|
||||
# Delay queue processing if a download_delay is configured
|
||||
now = time()
|
||||
if site.download_delay:
|
||||
penalty = site.download_delay - now + site.lastseen
|
||||
delay = site.download_delay()
|
||||
if delay:
|
||||
penalty = delay - now + site.lastseen
|
||||
if penalty > 0:
|
||||
d = defer.Deferred()
|
||||
d.addCallback(self._process_queue)
|
||||
|
|
|
|||
|
|
@ -1,66 +0,0 @@
|
|||
"""
|
||||
Crawler class
|
||||
|
||||
The Crawler class can be used to crawl pages using the Scrapy crawler from
|
||||
outside a Scrapy project, for example, from a standalone script.
|
||||
|
||||
To use it, instantiate it and call the "crawl" method with one (or more)
|
||||
requests. For example:
|
||||
|
||||
>>> from scrapy.crawler import Crawler
|
||||
>>> from scrapy.http import Request
|
||||
>>> def parse_response(response):
|
||||
... print "Visited: %s" % response.url
|
||||
...
|
||||
>>> request = Request('http://scrapy.org', callback=parse_response)
|
||||
>>> crawler = Crawler()
|
||||
>>> crawler.crawl(request)
|
||||
Visited: http://scrapy.org
|
||||
>>>
|
||||
|
||||
Request callbacks follow the same API of spiders callback, which means that all
|
||||
requests returned from the callbacks will be followed.
|
||||
|
||||
See examples/scripts/count_and_follow_links.py for a more detailed example.
|
||||
|
||||
WARNING: The Crawler class currently has a big limitation - it cannot be used
|
||||
more than once in the same Python process. This is due to the fact that Twisted
|
||||
reactors cannot be restarted. Hopefully, this limitation will be removed in the
|
||||
future.
|
||||
"""
|
||||
|
||||
from scrapy.xlib.pydispatch import dispatcher
|
||||
from scrapy.core.manager import scrapymanager
|
||||
from scrapy.core.engine import scrapyengine
|
||||
from scrapy.conf import settings as scrapy_settings
|
||||
from scrapy import log
|
||||
|
||||
class Crawler(object):
|
||||
|
||||
def __init__(self, enable_log=False, stop_on_error=False, silence_errors=False, \
|
||||
settings=None):
|
||||
self.stop_on_error = stop_on_error
|
||||
self.silence_errors = silence_errors
|
||||
# disable offsite middleware (by default) because it prevents free crawling
|
||||
if settings is not None:
|
||||
settings.overrides.update(settings)
|
||||
scrapy_settings.overrides['SPIDER_MIDDLEWARES'] = {
|
||||
'scrapy.contrib.spidermiddleware.offsite.OffsiteMiddleware': None}
|
||||
scrapy_settings.overrides['LOG_ENABLED'] = enable_log
|
||||
scrapymanager.configure()
|
||||
dispatcher.connect(self._logmessage_received, signal=log.logmessage_received)
|
||||
|
||||
def crawl(self, *args):
|
||||
scrapymanager.runonce(*args)
|
||||
|
||||
def stop(self):
|
||||
scrapyengine.stop()
|
||||
log.log_level = log.SILENT
|
||||
scrapyengine.kill()
|
||||
|
||||
def _logmessage_received(self, message, level):
|
||||
if level <= log.ERROR:
|
||||
if not self.silence_errors:
|
||||
print "Crawler error: %s" % message
|
||||
if self.stop_on_error:
|
||||
self.stop()
|
||||
|
|
@ -23,9 +23,6 @@ class HtmlResponse(TextResponse):
|
|||
METATAG_RE = re.compile(r'<meta\s+%s\s+%s' % (_httpequiv_re, _content_re), re.I)
|
||||
METATAG_RE2 = re.compile(r'<meta\s+%s\s+%s' % (_content_re, _httpequiv_re), re.I)
|
||||
|
||||
def body_encoding(self):
|
||||
return self._body_declared_encoding() or super(HtmlResponse, self).body_encoding()
|
||||
|
||||
@memoizemethod_noargs
|
||||
def _body_declared_encoding(self):
|
||||
chunk = self.body[:5000]
|
||||
|
|
|
|||
|
|
@ -11,6 +11,7 @@ from scrapy.xlib.BeautifulSoup import UnicodeDammit
|
|||
|
||||
from scrapy.http.response import Response
|
||||
from scrapy.utils.python import memoizemethod_noargs
|
||||
from scrapy.utils.encoding import encoding_exists, resolve_encoding
|
||||
from scrapy.conf import settings
|
||||
|
||||
class TextResponse(Response):
|
||||
|
|
@ -18,12 +19,12 @@ class TextResponse(Response):
|
|||
_DEFAULT_ENCODING = settings['DEFAULT_RESPONSE_ENCODING']
|
||||
_ENCODING_RE = re.compile(r'charset=([\w-]+)', re.I)
|
||||
|
||||
__slots__ = ['_encoding', '_body_inferred_encoding']
|
||||
__slots__ = ['_encoding', '_cached_benc']
|
||||
|
||||
def __init__(self, url, status=200, headers=None, body=None, meta=None, \
|
||||
flags=None, encoding=None):
|
||||
self._encoding = encoding
|
||||
self._body_inferred_encoding = None
|
||||
self._cached_benc = None
|
||||
super(TextResponse, self).__init__(url, status, headers, body, meta, flags)
|
||||
|
||||
def _get_url(self):
|
||||
|
|
@ -56,31 +57,39 @@ class TextResponse(Response):
|
|||
|
||||
@property
|
||||
def encoding(self):
|
||||
return self._encoding or self.headers_encoding() or self.body_encoding()
|
||||
enc = self._declared_encoding()
|
||||
if not (enc and encoding_exists(enc)):
|
||||
enc = self._body_inferred_encoding() or self._DEFAULT_ENCODING
|
||||
return resolve_encoding(enc)
|
||||
|
||||
@memoizemethod_noargs
|
||||
def headers_encoding(self):
|
||||
content_type = self.headers.get('Content-Type')
|
||||
if content_type:
|
||||
encoding = self._ENCODING_RE.search(content_type)
|
||||
if encoding:
|
||||
return encoding.group(1)
|
||||
def _declared_encoding(self):
|
||||
return self._encoding or self._headers_encoding() \
|
||||
or self._body_declared_encoding()
|
||||
|
||||
@memoizemethod_noargs
|
||||
def body_as_unicode(self):
|
||||
"""Return body as unicode"""
|
||||
possible_encodings = (self._encoding, self.headers_encoding(), \
|
||||
self._body_declared_encoding())
|
||||
dammit = UnicodeDammit(self.body, possible_encodings)
|
||||
self._body_inferred_encoding = dammit.originalEncoding
|
||||
if self._body_inferred_encoding in ('ascii', None):
|
||||
self._body_inferred_encoding = self._DEFAULT_ENCODING
|
||||
return dammit.unicode
|
||||
denc = self._declared_encoding()
|
||||
dencs = [resolve_encoding(denc)] if denc else []
|
||||
dammit = UnicodeDammit(self.body, dencs)
|
||||
benc = dammit.originalEncoding
|
||||
self._cached_benc = benc if benc != 'ascii' else None
|
||||
return self.body.decode(benc) if benc == 'utf-16' else dammit.unicode
|
||||
|
||||
def body_encoding(self):
|
||||
if self._body_inferred_encoding is None:
|
||||
@memoizemethod_noargs
|
||||
def _headers_encoding(self):
|
||||
content_type = self.headers.get('Content-Type')
|
||||
if content_type:
|
||||
m = self._ENCODING_RE.search(content_type)
|
||||
if m:
|
||||
encoding = m.group(1)
|
||||
if encoding_exists(encoding):
|
||||
return encoding
|
||||
|
||||
def _body_inferred_encoding(self):
|
||||
if self._cached_benc is None:
|
||||
self.body_as_unicode()
|
||||
return self._body_inferred_encoding
|
||||
return self._cached_benc
|
||||
|
||||
def _body_declared_encoding(self):
|
||||
# implemented in subclasses (XmlResponse, HtmlResponse)
|
||||
|
|
|
|||
|
|
@ -18,9 +18,6 @@ class XmlResponse(TextResponse):
|
|||
_encoding_re = _template % ('encoding', r'(?P<charset>[\w-]+)')
|
||||
XMLDECL_RE = re.compile(r'<\?xml\s.*?%s' % _encoding_re, re.I)
|
||||
|
||||
def body_encoding(self):
|
||||
return self._body_declared_encoding() or super(XmlResponse, self).body_encoding()
|
||||
|
||||
@memoizemethod_noargs
|
||||
def _body_declared_encoding(self):
|
||||
chunk = self.body[:5000]
|
||||
|
|
|
|||
|
|
@ -29,8 +29,9 @@ BOT_NAME = settings['BOT_NAME']
|
|||
# args: message, level, spider
|
||||
logmessage_received = object()
|
||||
|
||||
# default logging level
|
||||
# default values
|
||||
log_level = DEBUG
|
||||
log_encoding = 'utf-8'
|
||||
|
||||
started = False
|
||||
|
||||
|
|
@ -47,11 +48,12 @@ def _get_log_level(level_name_or_id=None):
|
|||
|
||||
def start(logfile=None, loglevel=None, logstdout=None):
|
||||
"""Initialize and start logging facility"""
|
||||
global log_level, started
|
||||
global log_level, log_encoding, started
|
||||
|
||||
if started or not settings.getbool('LOG_ENABLED'):
|
||||
return
|
||||
log_level = _get_log_level(loglevel)
|
||||
log_encoding = settings['LOG_ENCODING']
|
||||
started = True
|
||||
|
||||
# set log observer
|
||||
|
|
@ -74,7 +76,7 @@ def msg(message, level=INFO, component=BOT_NAME, domain=None, spider=None):
|
|||
dispatcher.send(signal=logmessage_received, message=message, level=level, \
|
||||
spider=spider)
|
||||
system = domain or (spider.domain_name if spider else component)
|
||||
msg_txt = unicode_to_str("%s: %s" % (level_names[level], message))
|
||||
msg_txt = unicode_to_str("%s: %s" % (level_names[level], message), log_encoding)
|
||||
log.msg(msg_txt, system=system)
|
||||
|
||||
def exc(message, level=ERROR, component=BOT_NAME, domain=None, spider=None):
|
||||
|
|
@ -93,5 +95,5 @@ def err(_stuff=None, _why=None, **kwargs):
|
|||
"use 'spider' argument instead", DeprecationWarning, stacklevel=2)
|
||||
kwargs['system'] = domain or (spider.domain_name if spider else component)
|
||||
if _why:
|
||||
_why = unicode_to_str("ERROR: %s" % _why)
|
||||
_why = unicode_to_str("ERROR: %s" % _why, log_encoding)
|
||||
log.err(_stuff, _why, **kwargs)
|
||||
|
|
|
|||
|
|
@ -47,34 +47,26 @@ class MailSender(object):
|
|||
part = MIMEBase(*mimetype.split('/'))
|
||||
part.set_payload(f.read())
|
||||
Encoders.encode_base64(part)
|
||||
part.add_header('Content-Disposition', 'attachment; filename="%s"' % attach_name)
|
||||
part.add_header('Content-Disposition', 'attachment; filename="%s"' \
|
||||
% attach_name)
|
||||
msg.attach(part)
|
||||
else:
|
||||
msg.set_payload(body)
|
||||
|
||||
# FIXME ---------------------------------------------------------------------
|
||||
# There seems to be a problem with sending emails using deferreds when
|
||||
# the last thing left to do is sending the mail, cause the engine stops
|
||||
# the reactor and the email don't get send. we need to fix this. until
|
||||
# then, we'll revert to use Python standard (IO-blocking) smtplib.
|
||||
|
||||
#dfd = self._sendmail(self.smtphost, self.mailfrom, rcpts, msg.as_string())
|
||||
#dfd.addCallbacks(self._sent_ok, self._sent_failed,
|
||||
# callbackArgs=[to, cc, subject, len(attachs)],
|
||||
# errbackArgs=[to, cc, subject, len(attachs)])
|
||||
import smtplib
|
||||
smtp = smtplib.SMTP(self.smtphost)
|
||||
smtp.sendmail(self.mailfrom, rcpts, msg.as_string())
|
||||
log.msg('Mail sent: To=%s Cc=%s Subject="%s"' % (to, cc, subject))
|
||||
smtp.close()
|
||||
# ---------------------------------------------------------------------------
|
||||
dfd = self._sendmail(self.smtphost, self.mailfrom, rcpts, msg.as_string())
|
||||
dfd.addCallbacks(self._sent_ok, self._sent_failed,
|
||||
callbackArgs=[to, cc, subject, len(attachs)],
|
||||
errbackArgs=[to, cc, subject, len(attachs)])
|
||||
reactor.addSystemEventTrigger('before', 'shutdown', lambda: dfd)
|
||||
|
||||
def _sent_ok(self, result, to, cc, subject, nattachs):
|
||||
log.msg('Mail sent OK: To=%s Cc=%s Subject="%s" Attachs=%d' % (to, cc, subject, nattachs))
|
||||
log.msg('Mail sent OK: To=%s Cc=%s Subject="%s" Attachs=%d' % \
|
||||
(to, cc, subject, nattachs))
|
||||
|
||||
def _sent_failed(self, failure, to, cc, subject, nattachs):
|
||||
errstr = str(failure.value)
|
||||
log.msg('Unable to send mail: To=%s Cc=%s Subject="%s" Attachs=%d - %s' % (to, cc, subject, nattachs, errstr), level=log.ERROR)
|
||||
log.msg('Unable to send mail: To=%s Cc=%s Subject="%s" Attachs=%d - %s' % \
|
||||
(to, cc, subject, nattachs, errstr), level=log.ERROR)
|
||||
|
||||
def _sendmail(self, smtphost, from_addr, to_addrs, msg, port=25):
|
||||
""" This is based on twisted.mail.smtp.sendmail except that it
|
||||
|
|
|
|||
|
|
@ -29,8 +29,8 @@ class XPathSelector(object_ref):
|
|||
self.doc = Libxml2Document(response, factory=self._get_libxml2_doc)
|
||||
self.xmlNode = self.doc.xmlDoc
|
||||
elif text:
|
||||
response = TextResponse(url='about:blank', body=unicode_to_str(text), \
|
||||
encoding='utf-8')
|
||||
response = TextResponse(url='about:blank', \
|
||||
body=unicode_to_str(text, 'utf-8'), encoding='utf-8')
|
||||
self.doc = Libxml2Document(response, factory=self._get_libxml2_doc)
|
||||
self.xmlNode = self.doc.xmlDoc
|
||||
self.expr = expr
|
||||
|
|
|
|||
|
|
@ -4,5 +4,5 @@
|
|||
# See: http://doc.scrapy.org/topics/item-pipeline.html
|
||||
|
||||
class ${ProjectName}Pipeline(object):
|
||||
def process_item(self, domain, item):
|
||||
def process_item(self, spider, item):
|
||||
return item
|
||||
|
|
|
|||
|
|
@ -10,15 +10,15 @@ class $classname(CrawlSpider):
|
|||
start_urls = ['http://www.$site/']
|
||||
|
||||
rules = (
|
||||
Rule(SgmlLinkExtractor(allow=(r'Items/', )), 'parse_item', follow=True),
|
||||
Rule(SgmlLinkExtractor(allow=r'Items/'), callback='parse_item', follow=True),
|
||||
)
|
||||
|
||||
def parse_item(self, response):
|
||||
xs = HtmlXPathSelector(response)
|
||||
hxs = HtmlXPathSelector(response)
|
||||
i = ${ProjectName}Item()
|
||||
#i['site_id'] = xs.select('//input[@id="sid"]/@value').extract()
|
||||
#i['name'] = xs.select('//div[@id="name"]').extract()
|
||||
#i['description'] = xs.select('//div[@id="description"]').extract()
|
||||
#i['site_id'] = hxs.select('//input[@id="sid"]/@value').extract()
|
||||
#i['name'] = hxs.select('//div[@id="name"]').extract()
|
||||
#i['description'] = hxs.select('//div[@id="description"]').extract()
|
||||
return i
|
||||
|
||||
SPIDER = $classname()
|
||||
|
|
|
|||
|
|
@ -0,0 +1,94 @@
|
|||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.http import Request
|
||||
from scrapy.http import Response
|
||||
|
||||
from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher
|
||||
|
||||
import re
|
||||
|
||||
class MatchersTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
pass
|
||||
|
||||
def test_base_matcher(self):
|
||||
matcher = BaseMatcher()
|
||||
|
||||
request = Request('http://example.com')
|
||||
response = Response('http://example.com')
|
||||
|
||||
self.assertTrue(matcher.matches_request(request))
|
||||
self.assertTrue(matcher.matches_response(response))
|
||||
|
||||
def test_url_matcher(self):
|
||||
matcher = UrlMatcher('http://example.com')
|
||||
|
||||
request = Request('http://example.com')
|
||||
response = Response('http://example.com')
|
||||
|
||||
self.failUnless(matcher.matches_request(request))
|
||||
self.failUnless(matcher.matches_request(response))
|
||||
|
||||
request = Request('http://example2.com')
|
||||
response = Response('http://example2.com')
|
||||
|
||||
self.failIf(matcher.matches_request(request))
|
||||
self.failIf(matcher.matches_request(response))
|
||||
|
||||
def test_url_regex_matcher(self):
|
||||
matcher = UrlRegexMatcher(r'sample')
|
||||
urls = (
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/sample3.html',
|
||||
'http://example.com/sample4.html',
|
||||
)
|
||||
for url in urls:
|
||||
request, response = Request(url), Response(url)
|
||||
self.failUnless(matcher.matches_request(request))
|
||||
self.failUnless(matcher.matches_response(response))
|
||||
|
||||
matcher = UrlRegexMatcher(r'sample_fail')
|
||||
for url in urls:
|
||||
request, response = Request(url), Response(url)
|
||||
self.failIf(matcher.matches_request(request))
|
||||
self.failIf(matcher.matches_response(response))
|
||||
|
||||
matcher = UrlRegexMatcher(r'SAMPLE\d+', re.IGNORECASE)
|
||||
for url in urls:
|
||||
request, response = Request(url), Response(url)
|
||||
self.failUnless(matcher.matches_request(request))
|
||||
self.failUnless(matcher.matches_response(response))
|
||||
|
||||
def test_url_list_matcher(self):
|
||||
urls = (
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/sample3.html',
|
||||
'http://example.com/sample4.html',
|
||||
)
|
||||
urls2 = (
|
||||
'http://example.com/sample5.html',
|
||||
'http://example.com/sample6.html',
|
||||
'http://example.com/sample7.html',
|
||||
'http://example.com/sample8.html',
|
||||
'http://example.com/',
|
||||
)
|
||||
matcher = UrlListMatcher(urls)
|
||||
|
||||
# match urls
|
||||
for url in urls:
|
||||
request, response = Request(url), Response(url)
|
||||
self.failUnless(matcher.matches_request(request))
|
||||
self.failUnless(matcher.matches_response(response))
|
||||
|
||||
# non-match urls
|
||||
for url in urls2:
|
||||
request, response = Request(url), Response(url)
|
||||
self.failIf(matcher.matches_request(request))
|
||||
self.failIf(matcher.matches_response(response))
|
||||
|
||||
|
|
@ -0,0 +1,156 @@
|
|||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.http import Request
|
||||
from scrapy.http import HtmlResponse
|
||||
from scrapy.tests import get_testdata
|
||||
|
||||
from scrapy.contrib_exp.crawlspider.reqext import BaseSgmlRequestExtractor
|
||||
from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
|
||||
from scrapy.contrib_exp.crawlspider.reqext import XPathRequestExtractor
|
||||
|
||||
class AbstractRequestExtractorTest(unittest.TestCase):
|
||||
|
||||
def _requests_equals(self, list1, list2):
|
||||
"""Compares request's urls and link_text"""
|
||||
for (r1, r2) in zip(list1, list2):
|
||||
if r1.url != r2.url:
|
||||
return False
|
||||
if r1.meta['link_text'] != r2.meta['link_text']:
|
||||
return False
|
||||
# all equal
|
||||
return True
|
||||
|
||||
|
||||
class RequestExtractorTest(AbstractRequestExtractorTest):
|
||||
|
||||
def test_basic(self):
|
||||
base_url = 'http://example.org/somepage/index.html'
|
||||
html = """<html><head><title>Page title<title>
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
<p><a href="/about.html">About us</a></p>
|
||||
<img src="/logo.png" alt="Company logo (not a link)" />
|
||||
<p><a href="../othercat.html">Other category</a></p>
|
||||
<p><a href="/" /></p></body></html>"""
|
||||
requests = [
|
||||
Request('http://example.org/somepage/item/12.html',
|
||||
meta={'link_text': 'Item 12'}),
|
||||
Request('http://example.org/about.html',
|
||||
meta={'link_text': 'About us'}),
|
||||
Request('http://example.org/othercat.html',
|
||||
meta={'link_text': 'Other category'}),
|
||||
Request('http://example.org/',
|
||||
meta={'link_text': ''}),
|
||||
]
|
||||
|
||||
response = HtmlResponse(base_url, body=html)
|
||||
reqx = BaseSgmlRequestExtractor() # default: tag=a, attr=href
|
||||
|
||||
self.failUnless(
|
||||
self._requests_equals(requests, reqx.extract_requests(response))
|
||||
)
|
||||
|
||||
def test_base_url(self):
|
||||
reqx = BaseSgmlRequestExtractor()
|
||||
|
||||
html = """<html><head><title>Page title<title>
|
||||
<base href="http://otherdomain.com/base/" />
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
</body></html>"""
|
||||
response = HtmlResponse("https://example.org/p/index.html", body=html)
|
||||
reqs = reqx.extract_requests(response)
|
||||
self.failUnless(self._requests_equals( \
|
||||
[Request('http://otherdomain.com/base/item/12.html', \
|
||||
meta={'link_text': 'Item 12'})], reqs), reqs)
|
||||
|
||||
# base url is an absolute path and relative to host
|
||||
html = """<html><head><title>Page title<title>
|
||||
<base href="/" />
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
</body></html>"""
|
||||
response = HtmlResponse("https://example.org/p/index.html", body=html)
|
||||
reqs = reqx.extract_requests(response)
|
||||
self.failUnless(self._requests_equals( \
|
||||
[Request('https://example.org/item/12.html', \
|
||||
meta={'link_text': 'Item 12'})], reqs), reqs)
|
||||
|
||||
# base url has no scheme
|
||||
html = """<html><head><title>Page title<title>
|
||||
<base href="//noscheme.com/base/" />
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
</body></html>"""
|
||||
response = HtmlResponse("https://example.org/p/index.html", body=html)
|
||||
reqs = reqx.extract_requests(response)
|
||||
self.failUnless(self._requests_equals( \
|
||||
[Request('https://noscheme.com/base/item/12.html', \
|
||||
meta={'link_text': 'Item 12'})], reqs), reqs)
|
||||
|
||||
def test_extraction_encoding(self):
|
||||
#TODO: use own fixtures
|
||||
body = get_testdata('link_extractor', 'linkextractor_noenc.html')
|
||||
response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body,
|
||||
headers={'Content-Type': ['text/html; charset=utf-8']})
|
||||
response_noenc = HtmlResponse(url='http://example.com/noenc',
|
||||
body=body)
|
||||
body = get_testdata('link_extractor', 'linkextractor_latin1.html')
|
||||
response_latin1 = HtmlResponse(url='http://example.com/latin1',
|
||||
body=body)
|
||||
|
||||
reqx = BaseSgmlRequestExtractor()
|
||||
self.failUnless(
|
||||
self._requests_equals(
|
||||
reqx.extract_requests(response_utf8),
|
||||
[ Request(url='http://example.com/sample_%C3%B1.html',
|
||||
meta={'link_text': ''}),
|
||||
Request(url='http://example.com/sample_%E2%82%AC.html',
|
||||
meta={'link_text':
|
||||
'sample \xe2\x82\xac text'.decode('utf-8')}) ]
|
||||
)
|
||||
)
|
||||
|
||||
self.failUnless(
|
||||
self._requests_equals(
|
||||
reqx.extract_requests(response_noenc),
|
||||
[ Request(url='http://example.com/sample_%C3%B1.html',
|
||||
meta={'link_text': ''}),
|
||||
Request(url='http://example.com/sample_%E2%82%AC.html',
|
||||
meta={'link_text':
|
||||
'sample \xe2\x82\xac text'.decode('utf-8')}) ]
|
||||
)
|
||||
)
|
||||
|
||||
self.failUnless(
|
||||
self._requests_equals(
|
||||
reqx.extract_requests(response_latin1),
|
||||
[ Request(url='http://example.com/sample_%F1.html',
|
||||
meta={'link_text': ''}),
|
||||
Request(url='http://example.com/sample_%E1.html',
|
||||
meta={'link_text':
|
||||
'sample \xe1 text'.decode('latin1')}) ]
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
class SgmlRequestExtractorTest(AbstractRequestExtractorTest):
|
||||
pass
|
||||
|
||||
|
||||
class XPathRequestExtractorTest(AbstractRequestExtractorTest):
|
||||
|
||||
def setUp(self):
|
||||
# TODO: use own fixtures
|
||||
body = get_testdata('link_extractor', 'sgml_linkextractor.html')
|
||||
self.response = HtmlResponse(url='http://example.com/index', body=body)
|
||||
|
||||
|
||||
def test_restrict_xpaths(self):
|
||||
reqx = XPathRequestExtractor('//div[@id="subwrapper"]')
|
||||
self.failUnless(
|
||||
self._requests_equals(
|
||||
reqx.extract_requests(self.response),
|
||||
[ Request(url='http://example.com/sample1.html',
|
||||
meta={'link_text': ''}),
|
||||
Request(url='http://example.com/sample2.html',
|
||||
meta={'link_text': 'sample 2'}) ]
|
||||
)
|
||||
)
|
||||
|
||||
|
|
@ -0,0 +1,128 @@
|
|||
from twisted.internet import defer
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.http import Request
|
||||
from scrapy.http import HtmlResponse
|
||||
from scrapy.utils.python import equal_attributes
|
||||
|
||||
from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
|
||||
from scrapy.contrib_exp.crawlspider.reqgen import RequestGenerator
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
|
||||
|
||||
class RequestGeneratorTest(unittest.TestCase):
|
||||
|
||||
def setUp(self):
|
||||
url = 'http://example.org/somepage/index.html'
|
||||
html = """<html><head><title>Page title<title>
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
<p><a href="/about.html">About us</a></p>
|
||||
<img src="/logo.png" alt="Company logo (not a link)" />
|
||||
<p><a href="../othercat.html">Other category</a></p>
|
||||
<p><a href="/" /></p></body></html>"""
|
||||
|
||||
self.response = HtmlResponse(url, body=html)
|
||||
self.deferred = defer.Deferred()
|
||||
self.requests = [
|
||||
Request('http://example.org/somepage/item/12.html',
|
||||
meta={'link_text': 'Item 12'}),
|
||||
Request('http://example.org/about.html',
|
||||
meta={'link_text': 'About us'}),
|
||||
Request('http://example.org/othercat.html',
|
||||
meta={'link_text': 'Other category'}),
|
||||
Request('http://example.org/',
|
||||
meta={'link_text': ''}),
|
||||
]
|
||||
|
||||
def _equal_requests_list(self, list1, list2):
|
||||
list1 = list(list1)
|
||||
list2 = list(list2)
|
||||
if not len(list1) == len(list2):
|
||||
return False
|
||||
|
||||
for (req1, req2) in zip(list1, list2):
|
||||
if not equal_attributes(req1, req2, ['url']):
|
||||
return False
|
||||
return True
|
||||
|
||||
def test_basic(self):
|
||||
reqgen = RequestGenerator([], [], callback=self.deferred)
|
||||
# returns generator
|
||||
requests = reqgen.generate_requests(self.response)
|
||||
self.failUnlessEqual(list(requests), [])
|
||||
|
||||
def test_request_extractor(self):
|
||||
extractors = [
|
||||
SgmlRequestExtractor()
|
||||
]
|
||||
|
||||
# extract all requests
|
||||
reqgen = RequestGenerator(extractors, [], callback=self.deferred)
|
||||
requests = reqgen.generate_requests(self.response)
|
||||
self.failUnless(self._equal_requests_list(requests, self.requests))
|
||||
|
||||
for req in requests:
|
||||
# check callback
|
||||
self.failUnlessEqual(req.deferred, self.deferred)
|
||||
|
||||
def test_request_processor(self):
|
||||
extractors = [
|
||||
SgmlRequestExtractor()
|
||||
]
|
||||
|
||||
processors = [
|
||||
Canonicalize(),
|
||||
FilterDupes(),
|
||||
]
|
||||
|
||||
reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
|
||||
requests = reqgen.generate_requests(self.response)
|
||||
self.failUnless(self._equal_requests_list(requests, self.requests))
|
||||
|
||||
# filter domain
|
||||
processors = [
|
||||
Canonicalize(),
|
||||
FilterDupes(),
|
||||
FilterDomain(deny='example.org'),
|
||||
]
|
||||
|
||||
reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
|
||||
requests = reqgen.generate_requests(self.response)
|
||||
self.failUnlessEqual(list(requests), [])
|
||||
|
||||
# filter url
|
||||
processors = [
|
||||
Canonicalize(),
|
||||
FilterDupes(),
|
||||
FilterUrl(deny=(r'about', r'othercat')),
|
||||
]
|
||||
|
||||
reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
|
||||
requests = reqgen.generate_requests(self.response)
|
||||
|
||||
self.failUnless(self._equal_requests_list(requests, [
|
||||
Request('http://example.org/somepage/item/12.html',
|
||||
meta={'link_text': 'Item 12'}),
|
||||
Request('http://example.org/',
|
||||
meta={'link_text': ''}),
|
||||
]))
|
||||
|
||||
processors = [
|
||||
Canonicalize(),
|
||||
FilterDupes(),
|
||||
FilterUrl(allow=r'/somepage/'),
|
||||
]
|
||||
|
||||
reqgen = RequestGenerator(extractors, processors, callback=self.deferred)
|
||||
requests = reqgen.generate_requests(self.response)
|
||||
|
||||
self.failUnless(self._equal_requests_list(requests, [
|
||||
Request('http://example.org/somepage/item/12.html',
|
||||
meta={'link_text': 'Item 12'}),
|
||||
]))
|
||||
|
||||
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,144 @@
|
|||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.http import Request
|
||||
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterDomain
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterUrl
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
|
||||
|
||||
import copy
|
||||
|
||||
class RequestProcessorsTest(unittest.TestCase):
|
||||
|
||||
def test_canonicalize_requests(self):
|
||||
urls = [
|
||||
'http://example.com/do?&b=1&a=2&c=3',
|
||||
'http://example.com/do?123,&q=a space',
|
||||
]
|
||||
urls_after = [
|
||||
'http://example.com/do?a=2&b=1&c=3',
|
||||
'http://example.com/do?123%2C=&q=a+space',
|
||||
]
|
||||
|
||||
proc = Canonicalize()
|
||||
results = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(results, urls_after)
|
||||
|
||||
def test_unique_requests(self):
|
||||
urls = [
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/sample3.html',
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
]
|
||||
urls_unique = [
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/sample3.html',
|
||||
]
|
||||
|
||||
proc = FilterDupes()
|
||||
results = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(results, urls_unique)
|
||||
|
||||
# Check custom attributes
|
||||
requests = [
|
||||
Request('http://example.com', method='GET'),
|
||||
Request('http://example.com', method='POST'),
|
||||
]
|
||||
proc = FilterDupes('url', 'method')
|
||||
self.failUnlessEqual(len(list(proc(requests))), 2)
|
||||
|
||||
proc = FilterDupes('url')
|
||||
self.failUnlessEqual(len(list(proc(requests))), 1)
|
||||
|
||||
def test_filter_domain(self):
|
||||
urls = [
|
||||
'http://blah1.com/index',
|
||||
'http://blah2.com/index',
|
||||
'http://blah1.com/section',
|
||||
'http://blah2.com/section',
|
||||
]
|
||||
|
||||
proc = FilterDomain(allow=('blah1.com'), deny=('blah2.com'))
|
||||
filtered = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(filtered, [
|
||||
'http://blah1.com/index',
|
||||
'http://blah1.com/section',
|
||||
])
|
||||
|
||||
proc = FilterDomain(deny=('blah1.com', 'blah2.com'))
|
||||
filtered = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(filtered, [])
|
||||
|
||||
proc = FilterDomain(allow=('blah1.com', 'blah2.com'))
|
||||
filtered = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(filtered, urls)
|
||||
|
||||
def test_filter_url(self):
|
||||
urls = [
|
||||
'http://blah1.com/index',
|
||||
'http://blah2.com/index',
|
||||
'http://blah1.com/section',
|
||||
'http://blah2.com/section',
|
||||
]
|
||||
|
||||
proc = FilterUrl(allow=(r'blah1'), deny=(r'blah2'))
|
||||
filtered = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(filtered, [
|
||||
'http://blah1.com/index',
|
||||
'http://blah1.com/section',
|
||||
])
|
||||
|
||||
proc = FilterUrl(deny=('blah1', 'blah2'))
|
||||
filtered = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(filtered, [])
|
||||
|
||||
proc = FilterUrl(allow=('index$', 'section$'))
|
||||
filtered = [req.url for req in proc(Request(url) for url in urls)]
|
||||
self.failUnlessEquals(filtered, urls)
|
||||
|
||||
|
||||
|
||||
def test_all_processors(self):
|
||||
urls = [
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/sample3.html',
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/do?&b=1&a=2&c=3',
|
||||
'http://example.com/do?123,&q=a space',
|
||||
]
|
||||
urls_processed = [
|
||||
'http://example.com/sample1.html',
|
||||
'http://example.com/sample2.html',
|
||||
'http://example.com/sample3.html',
|
||||
'http://example.com/do?a=2&b=1&c=3',
|
||||
'http://example.com/do?123%2C=&q=a+space',
|
||||
]
|
||||
|
||||
processors = [
|
||||
Canonicalize(),
|
||||
FilterDupes(),
|
||||
]
|
||||
|
||||
def _process(requests):
|
||||
"""Apply all processors"""
|
||||
# copy list
|
||||
processed = [copy.copy(req) for req in requests]
|
||||
for proc in processors:
|
||||
processed = proc(processed)
|
||||
return processed
|
||||
|
||||
# empty requests
|
||||
results1 = [r.url for r in _process([])]
|
||||
self.failUnlessEquals(results1, [])
|
||||
|
||||
# try urls
|
||||
requests = (Request(url) for url in urls)
|
||||
results2 = [r.url for r in _process(requests)]
|
||||
self.failUnlessEquals(results2, urls_processed)
|
||||
|
||||
|
|
@ -0,0 +1,262 @@
|
|||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.http import HtmlResponse
|
||||
from scrapy.spider import BaseSpider
|
||||
from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
|
||||
|
||||
from scrapy.contrib_exp.crawlspider.rules import CompiledRule
|
||||
from scrapy.contrib_exp.crawlspider.rules import Rule
|
||||
from scrapy.contrib_exp.crawlspider.rules import RulesManager
|
||||
|
||||
from functools import partial
|
||||
|
||||
class RuleInitializationTest(unittest.TestCase):
|
||||
|
||||
def test_fail_if_rule_null(self):
|
||||
# fail on empty rule
|
||||
self.failUnlessRaises(ValueError, Rule)
|
||||
self.failUnlessRaises(ValueError, Rule,
|
||||
**dict(callback=None, follow=None))
|
||||
self.failUnlessRaises(ValueError, Rule,
|
||||
**dict(callback=None, follow=False))
|
||||
|
||||
def test_minimal_arguments_to_instantiation(self):
|
||||
# not fail if callback set
|
||||
self.failUnless(Rule(callback=lambda: True))
|
||||
# not fail if follow set
|
||||
self.failUnless(Rule(follow=True))
|
||||
|
||||
def test_validate_default_attributes(self):
|
||||
# test null Rule
|
||||
rule = Rule(follow=True)
|
||||
self.failUnlessEqual(None, rule.matcher)
|
||||
self.failUnlessEqual(None, rule.callback)
|
||||
self.failUnlessEqual({}, rule.cb_kwargs)
|
||||
# follow default False
|
||||
self.failUnlessEqual(True, rule.follow)
|
||||
|
||||
def test_validate_attributes_set(self):
|
||||
matcher = BaseMatcher()
|
||||
callback = lambda: True
|
||||
rule = Rule(matcher, callback, True, a=1)
|
||||
# test attributes
|
||||
self.failUnlessEqual(matcher, rule.matcher)
|
||||
self.failUnlessEqual(callback, rule.callback)
|
||||
self.failUnlessEqual({'a': 1}, rule.cb_kwargs)
|
||||
self.failUnlessEqual(True, rule.follow)
|
||||
|
||||
class CompiledRuleInitializationTest(unittest.TestCase):
|
||||
|
||||
def test_fail_on_invalid_matcher(self):
|
||||
# pass with valid matcher
|
||||
self.failUnless(CompiledRule(BaseMatcher()),
|
||||
"Failed CompiledRule instantiation")
|
||||
|
||||
# at least needs valid matcher
|
||||
self.assertRaises(AssertionError, CompiledRule, None)
|
||||
self.assertRaises(AssertionError, CompiledRule, False)
|
||||
self.assertRaises(AssertionError, CompiledRule, True)
|
||||
|
||||
def test_fail_on_invalid_callback(self):
|
||||
# pass with valid callback
|
||||
callback = lambda: True
|
||||
self.failUnless(CompiledRule(BaseMatcher(), callback))
|
||||
# pass with callback none
|
||||
self.failUnless(CompiledRule(BaseMatcher(), None))
|
||||
|
||||
# assert on invalid callback
|
||||
self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
|
||||
'myfunc')
|
||||
|
||||
# numeric variable
|
||||
var = 123
|
||||
self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
|
||||
var)
|
||||
|
||||
class A:
|
||||
pass
|
||||
|
||||
# random instance
|
||||
self.assertRaises(AssertionError, CompiledRule, BaseMatcher(),
|
||||
A())
|
||||
|
||||
|
||||
def test_fail_on_invalid_follow_value(self):
|
||||
callback = lambda: True
|
||||
matcher = BaseMatcher()
|
||||
# pass bool
|
||||
self.failUnless(CompiledRule(matcher, callback, True))
|
||||
self.failUnless(CompiledRule(matcher, callback, False))
|
||||
|
||||
# assert with non-bool
|
||||
self.assertRaises(AssertionError, CompiledRule, matcher,
|
||||
callback, None)
|
||||
self.assertRaises(AssertionError, CompiledRule, matcher,
|
||||
callback, 1)
|
||||
|
||||
def test_validate_default_attributes(self):
|
||||
callback = lambda: True
|
||||
matcher = BaseMatcher()
|
||||
rule = CompiledRule(matcher, callback, True)
|
||||
|
||||
# test attributes
|
||||
self.failUnlessEqual(matcher, rule.matcher)
|
||||
self.failUnlessEqual(callback, rule.callback)
|
||||
self.failUnlessEqual(True, rule.follow)
|
||||
|
||||
|
||||
class RulesTest(unittest.TestCase):
|
||||
def test_rules_manager_basic(self):
|
||||
spider = BaseSpider()
|
||||
response1 = HtmlResponse('http://example.org')
|
||||
response2 = HtmlResponse('http://othersite.org')
|
||||
rulesman = RulesManager([], spider)
|
||||
|
||||
# should return none
|
||||
self.failIf(rulesman.get_rule_from_response(response1))
|
||||
self.failIf(rulesman.get_rule_from_response(response2))
|
||||
|
||||
# rules manager with match-all rule
|
||||
rulesman = RulesManager([
|
||||
Rule(BaseMatcher(), follow=True),
|
||||
], spider)
|
||||
|
||||
# returns CompiledRule
|
||||
rule1 = rulesman.get_rule_from_response(response1)
|
||||
rule2 = rulesman.get_rule_from_response(response2)
|
||||
|
||||
self.failUnless(isinstance(rule1, CompiledRule))
|
||||
self.failUnless(isinstance(rule2, CompiledRule))
|
||||
self.assert_(rule1 is rule2)
|
||||
self.failUnlessEqual(rule1.callback, None)
|
||||
self.failUnlessEqual(rule1.follow, True)
|
||||
|
||||
def test_rules_manager_empty_rule(self):
|
||||
spider = BaseSpider()
|
||||
response = HtmlResponse('http://example.org')
|
||||
|
||||
rulesman = RulesManager([Rule(follow=True)], spider)
|
||||
|
||||
rule = rulesman.get_rule_from_response(response)
|
||||
# default matcher if None: BaseMatcher
|
||||
self.failUnless(isinstance(rule.matcher, BaseMatcher))
|
||||
|
||||
def test_rules_manager_default_matcher(self):
|
||||
spider = BaseSpider()
|
||||
response = HtmlResponse('http://example.org')
|
||||
callback = lambda x: None
|
||||
|
||||
rulesman = RulesManager([
|
||||
Rule('http://example.org', callback),
|
||||
], spider, default_matcher=UrlMatcher)
|
||||
|
||||
rule = rulesman.get_rule_from_response(response)
|
||||
self.failUnless(isinstance(rule.matcher, UrlMatcher))
|
||||
|
||||
def test_rules_manager_matchers(self):
|
||||
spider = BaseSpider()
|
||||
response1 = HtmlResponse('http://example.org')
|
||||
response2 = HtmlResponse('http://othersite.org')
|
||||
|
||||
urlmatcher = UrlMatcher('http://example.org')
|
||||
basematcher = BaseMatcher()
|
||||
# callback needed for Rule
|
||||
callback = lambda x: None
|
||||
|
||||
# test fail matcher resolve
|
||||
self.assertRaises(ValueError, RulesManager,
|
||||
[Rule(False, callback)], spider)
|
||||
self.assertRaises(ValueError, RulesManager,
|
||||
[Rule(spider, callback)], spider)
|
||||
|
||||
rulesman = RulesManager([
|
||||
Rule(urlmatcher, callback),
|
||||
Rule(basematcher, callback),
|
||||
], spider)
|
||||
|
||||
# response1 matches example.org
|
||||
rule1 = rulesman.get_rule_from_response(response1)
|
||||
# response2 is catch by BaseMatcher()
|
||||
rule2 = rulesman.get_rule_from_response(response2)
|
||||
|
||||
self.failUnlessEqual(rule1.matcher, urlmatcher)
|
||||
self.failUnlessEqual(rule2.matcher, basematcher)
|
||||
|
||||
# reverse order. BaseMatcher should match all
|
||||
rulesman = RulesManager([
|
||||
Rule(basematcher, callback),
|
||||
Rule(urlmatcher, callback),
|
||||
], spider)
|
||||
|
||||
rule1 = rulesman.get_rule_from_response(response1)
|
||||
rule2 = rulesman.get_rule_from_response(response2)
|
||||
|
||||
self.failUnlessEqual(rule1.matcher, basematcher)
|
||||
self.failUnlessEqual(rule2.matcher, basematcher)
|
||||
self.failUnless(rule1 is rule2)
|
||||
|
||||
def test_rules_manager_callbacks(self):
|
||||
mycallback = lambda: True
|
||||
|
||||
spider = BaseSpider()
|
||||
spider.parse_item = lambda: True
|
||||
|
||||
response1 = HtmlResponse('http://example.org')
|
||||
response2 = HtmlResponse('http://othersite.org')
|
||||
|
||||
rulesman = RulesManager([
|
||||
Rule('example', mycallback),
|
||||
Rule('othersite', 'parse_item'),
|
||||
], spider, default_matcher=UrlRegexMatcher)
|
||||
|
||||
rule1 = rulesman.get_rule_from_response(response1)
|
||||
rule2 = rulesman.get_rule_from_response(response2)
|
||||
|
||||
self.failUnlessEqual(rule1.callback, mycallback)
|
||||
self.failUnlessEqual(rule2.callback, spider.parse_item)
|
||||
|
||||
# fail unknown callback
|
||||
self.assertRaises(AttributeError, RulesManager, [
|
||||
Rule(BaseMatcher(), 'mycallback')
|
||||
], spider)
|
||||
# fail not callable
|
||||
spider.not_callable = True
|
||||
self.assertRaises(AttributeError, RulesManager, [
|
||||
Rule(BaseMatcher(), 'not_callable')
|
||||
], spider)
|
||||
|
||||
|
||||
def test_rules_manager_callback_with_arguments(self):
|
||||
spider = BaseSpider()
|
||||
response = HtmlResponse('http://example.org')
|
||||
|
||||
kwargs = {'a': 1}
|
||||
|
||||
def myfunc(**mykwargs):
|
||||
return mykwargs
|
||||
|
||||
# verify return validation
|
||||
self.failUnlessEquals(kwargs, myfunc(**kwargs))
|
||||
|
||||
# test callback w/o arguments
|
||||
rulesman = RulesManager([
|
||||
Rule(BaseMatcher(), myfunc),
|
||||
], spider)
|
||||
rule = rulesman.get_rule_from_response(response)
|
||||
|
||||
# without arguments should return same callback
|
||||
self.failUnlessEqual(rule.callback, myfunc)
|
||||
|
||||
# test callback w/ arguments
|
||||
rulesman = RulesManager([
|
||||
Rule(BaseMatcher(), myfunc, **kwargs),
|
||||
], spider)
|
||||
rule = rulesman.get_rule_from_response(response)
|
||||
|
||||
# with argument should return partial applied callback
|
||||
self.failUnless(isinstance(rule.callback, partial))
|
||||
self.failUnlessEquals(kwargs, rule.callback())
|
||||
|
||||
|
||||
|
|
@ -0,0 +1,222 @@
|
|||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.http import Request
|
||||
from scrapy.http import HtmlResponse
|
||||
from scrapy.item import BaseItem
|
||||
from scrapy.utils.spider import iterate_spider_output
|
||||
|
||||
# basics
|
||||
from scrapy.contrib_exp.crawlspider import CrawlSpider
|
||||
from scrapy.contrib_exp.crawlspider import Rule
|
||||
|
||||
# matchers
|
||||
from scrapy.contrib_exp.crawlspider.matchers import BaseMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlRegexMatcher
|
||||
from scrapy.contrib_exp.crawlspider.matchers import UrlListMatcher
|
||||
|
||||
# extractors
|
||||
from scrapy.contrib_exp.crawlspider.reqext import SgmlRequestExtractor
|
||||
|
||||
# processors
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import Canonicalize
|
||||
from scrapy.contrib_exp.crawlspider.reqproc import FilterDupes
|
||||
|
||||
|
||||
# mock items
|
||||
class Item1(BaseItem):
|
||||
pass
|
||||
|
||||
class Item2(BaseItem):
|
||||
pass
|
||||
|
||||
class Item3(BaseItem):
|
||||
pass
|
||||
|
||||
|
||||
class CrawlSpiderTest(unittest.TestCase):
|
||||
|
||||
def spider_factory(self, rules=[],
|
||||
extractors=[], processors=[],
|
||||
start_urls=[]):
|
||||
# mock spider
|
||||
class Spider(CrawlSpider):
|
||||
def parse_item1(self, response):
|
||||
return Item1()
|
||||
|
||||
def parse_item2(self, response):
|
||||
return Item2()
|
||||
|
||||
def parse_item3(self, response):
|
||||
return Item3()
|
||||
|
||||
def parse_request1(self, response):
|
||||
return Request('http://example.org/request1')
|
||||
|
||||
def parse_request2(self, response):
|
||||
return Request('http://example.org/request2')
|
||||
|
||||
Spider.start_urls = start_urls
|
||||
Spider.rules = rules
|
||||
Spider.request_extractors = extractors
|
||||
Spider.request_processors = processors
|
||||
|
||||
return Spider()
|
||||
|
||||
def test_start_url_auto_rule(self):
|
||||
spider = self.spider_factory()
|
||||
# zero spider rules
|
||||
self.failUnlessEqual(len(spider.rules), 0)
|
||||
self.failUnlessEqual(len(spider._rulesman._rules), 0)
|
||||
|
||||
spider = self.spider_factory(start_urls=['http://example.org'])
|
||||
|
||||
self.failUnlessEqual(len(spider.rules), 0)
|
||||
self.failUnlessEqual(len(spider._rulesman._rules), 1)
|
||||
|
||||
def test_start_url_matcher(self):
|
||||
url = 'http://example.org'
|
||||
spider = self.spider_factory(start_urls=[url])
|
||||
|
||||
response = HtmlResponse(url)
|
||||
|
||||
rule = spider._rulesman.get_rule_from_response(response)
|
||||
self.failUnless(isinstance(rule.matcher, UrlListMatcher))
|
||||
|
||||
response = HtmlResponse(url + '/item.html')
|
||||
|
||||
rule = spider._rulesman.get_rule_from_response(response)
|
||||
self.failUnless(rule is None)
|
||||
|
||||
# TODO: remove this block
|
||||
# in previous version get_rule returns rule from response.request
|
||||
response.request = Request(url)
|
||||
rule = spider._rulesman.get_rule_from_response(response.request)
|
||||
self.failUnless(isinstance(rule.matcher, UrlListMatcher))
|
||||
self.failUnlessEqual(rule.follow, True)
|
||||
|
||||
def test_parse_callback(self):
|
||||
response = HtmlResponse('http://example.org')
|
||||
rules = (
|
||||
Rule(BaseMatcher(), 'parse_item1'),
|
||||
)
|
||||
spider = self.spider_factory(rules)
|
||||
|
||||
result = list(spider.parse(response))
|
||||
self.failUnlessEqual(len(result), 1)
|
||||
self.failUnless(isinstance(result[0], Item1))
|
||||
|
||||
def test_crawling_start_url(self):
|
||||
url = 'http://example.org/'
|
||||
html = """<html><head><title>Page title<title>
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
<p><a href="/about.html">About us</a></p>
|
||||
<img src="/logo.png" alt="Company logo (not a link)" />
|
||||
<p><a href="../othercat.html">Other category</a></p>
|
||||
<p><a href="/" /></p></body></html>"""
|
||||
response = HtmlResponse(url, body=html)
|
||||
|
||||
extractors = (SgmlRequestExtractor(), )
|
||||
spider = self.spider_factory(start_urls=[url],
|
||||
extractors=extractors)
|
||||
result = list(spider.parse(response))
|
||||
|
||||
# 1 request extracted: example.org/
|
||||
# because requests returns only matching
|
||||
self.failUnlessEqual(len(result), 1)
|
||||
|
||||
# we will add catch-all rule to extract all
|
||||
callback = lambda x: None
|
||||
rules = [Rule(r'\.html$', callback=callback)]
|
||||
spider = self.spider_factory(rules, start_urls=[url],
|
||||
extractors=extractors)
|
||||
result = list(spider.parse(response))
|
||||
|
||||
# 4 requests extracted
|
||||
# 3 of .html pattern
|
||||
# 1 of start url patter
|
||||
self.failUnlessEqual(len(result), 4)
|
||||
|
||||
def test_crawling_simple_rule(self):
|
||||
url = 'http://example.org/somepage/index.html'
|
||||
html = """<html><head><title>Page title<title>
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
<p><a href="/about.html">About us</a></p>
|
||||
<img src="/logo.png" alt="Company logo (not a link)" />
|
||||
<p><a href="../othercat.html">Other category</a></p>
|
||||
<p><a href="/" /></p></body></html>"""
|
||||
|
||||
response = HtmlResponse(url, body=html)
|
||||
|
||||
rules = (
|
||||
# first response callback
|
||||
Rule(r'index\.html', 'parse_item1'),
|
||||
)
|
||||
spider = self.spider_factory(rules)
|
||||
result = list(spider.parse(response))
|
||||
|
||||
# should return Item1
|
||||
self.failUnlessEqual(len(result), 1)
|
||||
self.failUnless(isinstance(result[0], Item1))
|
||||
|
||||
# test request generation
|
||||
rules = (
|
||||
# first response without callback and follow flag
|
||||
Rule(r'index\.html', follow=True),
|
||||
Rule(r'(\.html|/)$', 'parse_item1'),
|
||||
)
|
||||
spider = self.spider_factory(rules)
|
||||
result = list(spider.parse(response))
|
||||
|
||||
# 0 because spider does not have extractors
|
||||
self.failUnlessEqual(len(result), 0)
|
||||
|
||||
extractors = (SgmlRequestExtractor(), )
|
||||
|
||||
# instance spider with extractor
|
||||
spider = self.spider_factory(rules, extractors)
|
||||
result = list(spider.parse(response))
|
||||
# 4 requests extracted
|
||||
self.failUnlessEqual(len(result), 4)
|
||||
|
||||
def test_crawling_multiple_rules(self):
|
||||
html = """<html><head><title>Page title<title>
|
||||
<body><p><a href="item/12.html">Item 12</a></p>
|
||||
<p><a href="/about.html">About us</a></p>
|
||||
<img src="/logo.png" alt="Company logo (not a link)" />
|
||||
<p><a href="../othercat.html">Other category</a></p>
|
||||
<p><a href="/" /></p></body></html>"""
|
||||
|
||||
response = HtmlResponse('http://example.org/index.html', body=html)
|
||||
response1 = HtmlResponse('http://example.org/1.html')
|
||||
response2 = HtmlResponse('http://example.org/othercat.html')
|
||||
|
||||
rules = (
|
||||
Rule(r'\d+\.html$', 'parse_item1'),
|
||||
Rule(r'othercat\.html$', 'parse_item2'),
|
||||
# follow-only rules
|
||||
Rule(r'index\.html', 'parse_item3', follow=True)
|
||||
)
|
||||
extractors = [SgmlRequestExtractor()]
|
||||
spider = self.spider_factory(rules, extractors)
|
||||
|
||||
result = list(spider.parse(response))
|
||||
# 1 Item 2 Requests
|
||||
self.failUnlessEqual(len(result), 3)
|
||||
# parse_item3
|
||||
self.failUnless(isinstance(result[0], Item3))
|
||||
only_requests = lambda r: isinstance(r, Request)
|
||||
requests = filter(only_requests, result[1:])
|
||||
self.failUnlessEqual(len(requests), 2)
|
||||
self.failUnless(all(requests))
|
||||
|
||||
result1 = list(spider.parse(response1))
|
||||
# parse_item1
|
||||
self.failUnlessEqual(len(result1), 1)
|
||||
self.failUnless(isinstance(result1[0], Item1))
|
||||
|
||||
result2 = list(spider.parse(response2))
|
||||
# parse_item2
|
||||
self.failUnlessEqual(len(result2), 1)
|
||||
self.failUnless(isinstance(result2[0], Item2))
|
||||
|
||||
|
||||
|
|
@ -35,6 +35,20 @@ class LinkExtractorTestCase(unittest.TestCase):
|
|||
self.assertEqual(lx.extract_links(response),
|
||||
[Link(url='http://otherdomain.com/base/item/12.html', text='Item 12')])
|
||||
|
||||
# base url is an absolute path and relative to host
|
||||
html = """<html><head><title>Page title<title><base href="/" />
|
||||
<body><p><a href="item/12.html">Item 12</a></p></body></html>"""
|
||||
response = HtmlResponse("https://example.org/somepage/index.html", body=html)
|
||||
self.assertEqual(lx.extract_links(response),
|
||||
[Link(url='https://example.org/item/12.html', text='Item 12')])
|
||||
|
||||
# base url has no scheme
|
||||
html = """<html><head><title>Page title<title><base href="//noschemedomain.com/path/to/" />
|
||||
<body><p><a href="item/12.html">Item 12</a></p></body></html>"""
|
||||
response = HtmlResponse("https://example.org/somepage/index.html", body=html)
|
||||
self.assertEqual(lx.extract_links(response),
|
||||
[Link(url='https://noschemedomain.com/path/to/item/12.html', text='Item 12')])
|
||||
|
||||
def test_extraction_encoding(self):
|
||||
body = get_testdata('link_extractor', 'linkextractor_noenc.html')
|
||||
response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']})
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ import unittest
|
|||
from scrapy.contrib.downloadermiddleware.redirect import RedirectMiddleware
|
||||
from scrapy.spider import BaseSpider
|
||||
from scrapy.core.exceptions import IgnoreRequest
|
||||
from scrapy.http import Request, Response, Headers
|
||||
from scrapy.http import Request, Response, HtmlResponse, Headers
|
||||
|
||||
class RedirectMiddlewareTest(unittest.TestCase):
|
||||
|
||||
|
|
@ -58,7 +58,7 @@ class RedirectMiddlewareTest(unittest.TestCase):
|
|||
<head><meta http-equiv="refresh" content="5;url=http://example.org/newpage" /></head>
|
||||
</html>"""
|
||||
req = Request(url='http://example.org')
|
||||
rsp = Response(url='http://example.org', body=body)
|
||||
rsp = HtmlResponse(url='http://example.org', body=body)
|
||||
req2 = self.mw.process_response(req, rsp, self.spider)
|
||||
|
||||
assert isinstance(req2, Request)
|
||||
|
|
@ -70,7 +70,7 @@ class RedirectMiddlewareTest(unittest.TestCase):
|
|||
<head><meta http-equiv="refresh" content="1000;url=http://example.org/newpage" /></head>
|
||||
</html>"""
|
||||
req = Request(url='http://example.org')
|
||||
rsp = Response(url='http://example.org', body=body)
|
||||
rsp = HtmlResponse(url='http://example.org', body=body)
|
||||
rsp2 = self.mw.process_response(req, rsp, self.spider)
|
||||
|
||||
assert rsp is rsp2
|
||||
|
|
@ -81,7 +81,7 @@ class RedirectMiddlewareTest(unittest.TestCase):
|
|||
</html>"""
|
||||
req = Request(url='http://example.org', method='POST', body='test',
|
||||
headers={'Content-Type': 'text/plain', 'Content-length': '4'})
|
||||
rsp = Response(url='http://example.org', body=body)
|
||||
rsp = HtmlResponse(url='http://example.org', body=body)
|
||||
req2 = self.mw.process_response(req, rsp, self.spider)
|
||||
|
||||
assert isinstance(req2, Request)
|
||||
|
|
|
|||
|
|
@ -1,21 +0,0 @@
|
|||
import unittest
|
||||
|
||||
import scrapy # adds encoding aliases (if not added before)
|
||||
|
||||
class EncodingAliasesTestCase(unittest.TestCase):
|
||||
|
||||
def test_encoding_aliases(self):
|
||||
"""Test common encdoing aliases not included in Python"""
|
||||
|
||||
uni = u'\u041c\u041e\u0421K\u0412\u0410'
|
||||
str = uni.encode('windows-1251')
|
||||
self.assertEqual(uni.encode('windows-1251'), uni.encode('win-1251'))
|
||||
self.assertEqual(str.decode('windows-1251'), str.decode('win-1251'))
|
||||
|
||||
text = u'\u8f6f\u4ef6\u540d\u79f0'
|
||||
str = uni.encode('gb2312')
|
||||
self.assertEqual(uni.encode('gb2312'), uni.encode('zh-cn'))
|
||||
self.assertEqual(str.decode('gb2312'), str.decode('zh-cn'))
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
@ -2,6 +2,7 @@ import unittest
|
|||
import weakref
|
||||
|
||||
from scrapy.http import Response, TextResponse, HtmlResponse, XmlResponse, Headers
|
||||
from scrapy.utils.encoding import resolve_encoding
|
||||
from scrapy.conf import settings
|
||||
|
||||
|
||||
|
|
@ -112,10 +113,13 @@ class BaseResponseTest(unittest.TestCase):
|
|||
body_str = body
|
||||
|
||||
assert isinstance(response.body, str)
|
||||
self.assertEqual(response.encoding, encoding)
|
||||
self._assert_response_encoding(response, encoding)
|
||||
self.assertEqual(response.body, body_str)
|
||||
self.assertEqual(response.body_as_unicode(), body_unicode)
|
||||
|
||||
def _assert_response_encoding(self, response, encoding):
|
||||
self.assertEqual(response.encoding, resolve_encoding(encoding))
|
||||
|
||||
class ResponseText(BaseResponseTest):
|
||||
|
||||
def test_no_unicode_url(self):
|
||||
|
|
@ -134,14 +138,14 @@ class TextResponseTest(BaseResponseTest):
|
|||
|
||||
assert isinstance(r2, self.response_class)
|
||||
self.assertEqual(r2.url, "http://www.example.com/other")
|
||||
self.assertEqual(r2.encoding, "cp852")
|
||||
self._assert_response_encoding(r2, "cp852")
|
||||
self.assertEqual(r3.url, "http://www.example.com/other")
|
||||
self.assertEqual(r3.encoding, "latin1")
|
||||
self.assertEqual(r3._declared_encoding(), "latin1")
|
||||
|
||||
def test_unicode_url(self):
|
||||
# instantiate with unicode url without encoding (should set default encoding)
|
||||
resp = self.response_class(u"http://www.example.com/")
|
||||
self.assertEqual(resp.encoding, settings['DEFAULT_RESPONSE_ENCODING'])
|
||||
self._assert_response_encoding(resp, settings['DEFAULT_RESPONSE_ENCODING'])
|
||||
|
||||
# make sure urls are converted to str
|
||||
resp = self.response_class(url=u"http://www.example.com/", encoding='utf-8')
|
||||
|
|
@ -175,14 +179,18 @@ class TextResponseTest(BaseResponseTest):
|
|||
r2 = self.response_class("http://www.example.com", encoding='utf-8', body=u"\xa3")
|
||||
r3 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=iso-8859-1"]}, body="\xa3")
|
||||
r4 = self.response_class("http://www.example.com", body="\xa2\xa3")
|
||||
r5 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=None"]}, body="\xc2\xa3")
|
||||
|
||||
self.assertEqual(r1.headers_encoding(), "utf-8")
|
||||
self.assertEqual(r2.headers_encoding(), None)
|
||||
self.assertEqual(r2.encoding, 'utf-8')
|
||||
self.assertEqual(r3.headers_encoding(), "iso-8859-1")
|
||||
self.assertEqual(r3.encoding, 'iso-8859-1')
|
||||
self.assertEqual(r4.headers_encoding(), None)
|
||||
assert r4.body_encoding() is not None and r4.body_encoding() != 'ascii'
|
||||
self.assertEqual(r1._headers_encoding(), "utf-8")
|
||||
self.assertEqual(r2._headers_encoding(), None)
|
||||
self.assertEqual(r2._declared_encoding(), 'utf-8')
|
||||
self._assert_response_encoding(r2, 'utf-8')
|
||||
self.assertEqual(r3._headers_encoding(), "iso-8859-1")
|
||||
self.assertEqual(r3._declared_encoding(), "iso-8859-1")
|
||||
self.assertEqual(r4._headers_encoding(), None)
|
||||
self.assertEqual(r5._headers_encoding(), None)
|
||||
self._assert_response_encoding(r5, "utf-8")
|
||||
assert r4._body_inferred_encoding() is not None and r4._body_inferred_encoding() != 'ascii'
|
||||
self._assert_response_values(r1, 'utf-8', u"\xa3")
|
||||
self._assert_response_values(r2, 'utf-8', u"\xa3")
|
||||
self._assert_response_values(r3, 'iso-8859-1', u"\xa3")
|
||||
|
|
|
|||
|
|
@ -0,0 +1,25 @@
|
|||
import unittest
|
||||
|
||||
from scrapy.utils.encoding import encoding_exists, resolve_encoding
|
||||
|
||||
class UtilsEncodingTestCase(unittest.TestCase):
|
||||
|
||||
_ENCODING_ALIASES = {
|
||||
'foo': 'cp1252',
|
||||
'bar': 'none',
|
||||
}
|
||||
|
||||
def test_resolve_encoding(self):
|
||||
self.assertEqual(resolve_encoding('latin1', self._ENCODING_ALIASES),
|
||||
'latin1')
|
||||
self.assertEqual(resolve_encoding('foo', self._ENCODING_ALIASES),
|
||||
'cp1252')
|
||||
|
||||
def test_encoding_exists(self):
|
||||
assert encoding_exists('latin1', self._ENCODING_ALIASES)
|
||||
assert encoding_exists('foo', self._ENCODING_ALIASES)
|
||||
assert not encoding_exists('bar', self._ENCODING_ALIASES)
|
||||
assert not encoding_exists('none', self._ENCODING_ALIASES)
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
@ -1,7 +1,8 @@
|
|||
import operator
|
||||
import unittest
|
||||
|
||||
from scrapy.utils.python import str_to_unicode, unicode_to_str, \
|
||||
memoizemethod_noargs, isbinarytext
|
||||
memoizemethod_noargs, isbinarytext, equal_attributes
|
||||
|
||||
class UtilsPythonTestCase(unittest.TestCase):
|
||||
def test_str_to_unicode(self):
|
||||
|
|
@ -61,5 +62,52 @@ class UtilsPythonTestCase(unittest.TestCase):
|
|||
# finally some real binary bytes
|
||||
assert isbinarytext("\x02\xa3")
|
||||
|
||||
def test_equal_attributes(self):
|
||||
class Obj:
|
||||
pass
|
||||
|
||||
a = Obj()
|
||||
b = Obj()
|
||||
# no attributes given return False
|
||||
self.failIf(equal_attributes(a, b, []))
|
||||
# not existent attributes
|
||||
self.failIf(equal_attributes(a, b, ['x', 'y']))
|
||||
|
||||
a.x = 1
|
||||
b.x = 1
|
||||
# equal attribute
|
||||
self.failUnless(equal_attributes(a, b, ['x']))
|
||||
|
||||
b.y = 2
|
||||
# obj1 has no attribute y
|
||||
self.failIf(equal_attributes(a, b, ['x', 'y']))
|
||||
|
||||
a.y = 2
|
||||
# equal attributes
|
||||
self.failUnless(equal_attributes(a, b, ['x', 'y']))
|
||||
|
||||
a.y = 1
|
||||
# differente attributes
|
||||
self.failIf(equal_attributes(a, b, ['x', 'y']))
|
||||
|
||||
# test callable
|
||||
a.meta = {}
|
||||
b.meta = {}
|
||||
self.failUnless(equal_attributes(a, b, ['meta']))
|
||||
|
||||
# compare ['meta']['a']
|
||||
a.meta['z'] = 1
|
||||
b.meta['z'] = 1
|
||||
|
||||
get_z = operator.itemgetter('z')
|
||||
get_meta = operator.attrgetter('meta')
|
||||
compare_z = lambda obj: get_z(get_meta(obj))
|
||||
|
||||
self.failUnless(equal_attributes(a, b, [compare_z, 'x']))
|
||||
# fail z equality
|
||||
a.meta['z'] = 2
|
||||
self.failIf(equal_attributes(a, b, [compare_z, 'x']))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
import unittest
|
||||
import urlparse
|
||||
|
||||
from scrapy.xlib.BeautifulSoup import BeautifulSoup
|
||||
from scrapy.http import Response, TextResponse
|
||||
from scrapy.http import Response, TextResponse, HtmlResponse
|
||||
from scrapy.utils.response import body_or_str, get_base_url, get_meta_refresh, \
|
||||
response_httprepr, get_cached_beautifulsoup
|
||||
response_httprepr, get_cached_beautifulsoup, open_in_browser
|
||||
|
||||
class ResponseUtilsTest(unittest.TestCase):
|
||||
dummy_response = TextResponse(url='http://example.org/', body='dummy_response')
|
||||
|
|
@ -28,30 +29,46 @@ class ResponseUtilsTest(unittest.TestCase):
|
|||
self.assertTrue(isinstance(body_or_str(u'text', unicode=True), unicode))
|
||||
|
||||
def test_get_base_url(self):
|
||||
response = Response(url='http://example.org', body="""\
|
||||
response = HtmlResponse(url='https://example.org', body="""\
|
||||
<html>\
|
||||
<head><title>Dummy</title><base href='http://example.org/something' /></head>\
|
||||
<body>blahablsdfsal&</body>\
|
||||
</html>""")
|
||||
self.assertEqual(get_base_url(response), 'http://example.org/something')
|
||||
|
||||
# relative url with absolute path
|
||||
response = HtmlResponse(url='https://example.org', body="""\
|
||||
<html>\
|
||||
<head><title>Dummy</title><base href='/absolutepath' /></head>\
|
||||
<body>blahablsdfsal&</body>\
|
||||
</html>""")
|
||||
self.assertEqual(get_base_url(response), 'https://example.org/absolutepath')
|
||||
|
||||
# no scheme url
|
||||
response = HtmlResponse(url='https://example.org', body="""\
|
||||
<html>\
|
||||
<head><title>Dummy</title><base href='//noscheme.com/path' /></head>\
|
||||
<body>blahablsdfsal&</body>\
|
||||
</html>""")
|
||||
self.assertEqual(get_base_url(response), 'https://noscheme.com/path')
|
||||
|
||||
def test_get_meta_refresh(self):
|
||||
body = """
|
||||
<html>
|
||||
<head><title>Dummy</title><meta http-equiv="refresh" content="5;url=http://example.org/newpage" /></head>
|
||||
<body>blahablsdfsal&</body>
|
||||
</html>"""
|
||||
response = Response(url='http://example.org', body=body)
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (5, 'http://example.org/newpage'))
|
||||
|
||||
# refresh without url should return (None, None)
|
||||
body = """<meta http-equiv="refresh" content="5" />"""
|
||||
response = Response(url='http://example.org', body=body)
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
|
||||
body = """<meta http-equiv="refresh" content="5;
|
||||
url=http://example.org/newpage" /></head>"""
|
||||
response = Response(url='http://example.org', body=body)
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (5, 'http://example.org/newpage'))
|
||||
|
||||
# meta refresh in multiple lines
|
||||
|
|
@ -59,17 +76,17 @@ class ResponseUtilsTest(unittest.TestCase):
|
|||
<META
|
||||
HTTP-EQUIV="Refresh"
|
||||
CONTENT="1; URL=http://example.org/newpage">"""
|
||||
response = Response(url='http://example.org', body=body)
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (1, 'http://example.org/newpage'))
|
||||
|
||||
# entities in the redirect url
|
||||
body = """<meta http-equiv="refresh" content="3; url='http://www.example.com/other'">"""
|
||||
response = Response(url='http://example.com', body=body)
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://www.example.com/other'))
|
||||
|
||||
# relative redirects
|
||||
body = """<meta http-equiv="refresh" content="3; url=other.html">"""
|
||||
response = Response(url='http://example.com/page/this.html', body=body)
|
||||
response = TextResponse(url='http://example.com/page/this.html', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/page/other.html'))
|
||||
|
||||
# non-standard encodings (utf-16)
|
||||
|
|
@ -80,7 +97,7 @@ class ResponseUtilsTest(unittest.TestCase):
|
|||
|
||||
# non-ascii chars in the url (default encoding - utf8)
|
||||
body = """<meta http-equiv="refresh" content="3; url=http://example.com/to\xc2\xa3">"""
|
||||
response = Response(url='http://example.com', body=body)
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/to%C2%A3'))
|
||||
|
||||
# non-ascii chars in the url (custom encoding - latin1)
|
||||
|
|
@ -88,13 +105,8 @@ class ResponseUtilsTest(unittest.TestCase):
|
|||
response = TextResponse(url='http://example.com', body=body, encoding='latin1')
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/to%C2%A3'))
|
||||
|
||||
# wrong encodings (possibly caused by truncated chunks)
|
||||
body = """<meta http-equiv="refresh" content="3; url=http://example.com/this\xc2_THAT">"""
|
||||
response = Response(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/thisTHAT'))
|
||||
|
||||
# responses without refresh tag should return None None
|
||||
response = Response(url='http://example.org')
|
||||
response = TextResponse(url='http://example.org')
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
response = TextResponse(url='http://example.org')
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
|
|
@ -131,5 +143,18 @@ class ResponseUtilsTest(unittest.TestCase):
|
|||
assert soup1 is soup2
|
||||
assert soup1 is not soup3
|
||||
|
||||
def test_open_in_browser(self):
|
||||
url = "http:///www.example.com/some/page.html"
|
||||
body = "<html> <head> <title>test page</title> </head> <body>test body</body> </html>"
|
||||
def browser_open(burl):
|
||||
bbody = open(urlparse.urlparse(burl).path).read()
|
||||
assert '<base href="%s">' % url in bbody, "<base> tag not added"
|
||||
return True
|
||||
response = HtmlResponse(url, body=body)
|
||||
assert open_in_browser(response, _openfunc=browser_open), \
|
||||
"Browser not called"
|
||||
self.assertRaises(TypeError, open_in_browser, Response(url, body=body), \
|
||||
debug=True)
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
|
|
@ -1,11 +1,20 @@
|
|||
import codecs
|
||||
|
||||
def add_encoding_alias(encoding, alias, overwrite=False):
|
||||
from scrapy.conf import settings
|
||||
|
||||
_ENCODING_ALIASES = dict(settings['ENCODING_ALIASES_BASE'])
|
||||
_ENCODING_ALIASES.update(settings['ENCODING_ALIASES'])
|
||||
|
||||
def encoding_exists(encoding, _aliases=_ENCODING_ALIASES):
|
||||
"""Returns ``True`` if encoding is valid, otherwise returns ``False``"""
|
||||
try:
|
||||
codecs.lookup(alias)
|
||||
alias_exists = True
|
||||
codecs.lookup(resolve_encoding(encoding, _aliases))
|
||||
except LookupError:
|
||||
alias_exists = False
|
||||
if overwrite or not alias_exists:
|
||||
codec = codecs.lookup(encoding)
|
||||
codecs.register(lambda x: codec if x == alias else None)
|
||||
return False
|
||||
return True
|
||||
|
||||
def resolve_encoding(alias, _aliases=_ENCODING_ALIASES):
|
||||
"""Return the encoding the given alias maps to, or the alias as passed if
|
||||
no mapping is found.
|
||||
"""
|
||||
return _aliases.get(alias.lower(), alias)
|
||||
|
|
|
|||
|
|
@ -60,10 +60,10 @@ def remove_entities(text, keep=(), remove_illegal=True, encoding='utf-8'):
|
|||
|
||||
return _ent_re.sub(convert_entity, str_to_unicode(text, encoding))
|
||||
|
||||
def has_entities(text):
|
||||
return bool(_ent_re.search(str_to_unicode(text)))
|
||||
def has_entities(text, encoding=None):
|
||||
return bool(_ent_re.search(str_to_unicode(text, encoding)))
|
||||
|
||||
def replace_tags(text, token=''):
|
||||
def replace_tags(text, token='', encoding=None):
|
||||
"""Replace all markup tags found in the given text by the given token. By
|
||||
default token is a null string so it just remove all tags.
|
||||
|
||||
|
|
@ -71,43 +71,44 @@ def replace_tags(text, token=''):
|
|||
|
||||
Always returns a unicode string.
|
||||
"""
|
||||
return _tag_re.sub(token, str_to_unicode(text))
|
||||
return _tag_re.sub(token, str_to_unicode(text, encoding))
|
||||
|
||||
|
||||
def remove_comments(text):
|
||||
def remove_comments(text, encoding=None):
|
||||
""" Remove HTML Comments. """
|
||||
return re.sub('<!--.*?-->', u'', str_to_unicode(text), re.DOTALL)
|
||||
return re.sub('<!--.*?-->', u'', str_to_unicode(text, encoding), re.DOTALL)
|
||||
|
||||
def remove_tags(text, which_ones=()):
|
||||
def remove_tags(text, which_ones=(), encoding=None):
|
||||
""" Remove HTML Tags only.
|
||||
|
||||
which_ones -- is a tuple of which tags we want to remove.
|
||||
if is empty remove all tags.
|
||||
"""
|
||||
if which_ones:
|
||||
tags = ['<%s>|<%s .*?>|</%s>' % (tag,tag,tag) for tag in which_ones]
|
||||
tags = ['<%s>|<%s .*?>|</%s>' % (tag, tag, tag) for tag in which_ones]
|
||||
regex = '|'.join(tags)
|
||||
else:
|
||||
regex = '<.*?>'
|
||||
retags = re.compile(regex, re.DOTALL | re.IGNORECASE)
|
||||
|
||||
return retags.sub(u'', str_to_unicode(text))
|
||||
return retags.sub(u'', str_to_unicode(text, encoding))
|
||||
|
||||
def remove_tags_with_content(text, which_ones=()):
|
||||
def remove_tags_with_content(text, which_ones=(), encoding=None):
|
||||
""" Remove tags and its content.
|
||||
|
||||
which_ones -- is a tuple of which tags with its content we want to remove.
|
||||
if is empty do nothing.
|
||||
"""
|
||||
text = str_to_unicode(text)
|
||||
text = str_to_unicode(text, encoding)
|
||||
if which_ones:
|
||||
tags = '|'.join(['<%s.*?</%s>' % (tag,tag) for tag in which_ones])
|
||||
tags = '|'.join(['<%s.*?</%s>' % (tag, tag) for tag in which_ones])
|
||||
retags = re.compile(tags, re.DOTALL | re.IGNORECASE)
|
||||
text = retags.sub(u'', text)
|
||||
return text
|
||||
|
||||
|
||||
def replace_escape_chars(text, which_ones=('\n','\t','\r'), replace_by=u''):
|
||||
def replace_escape_chars(text, which_ones=('\n', '\t', '\r'), replace_by=u'', \
|
||||
encoding=None):
|
||||
""" Remove escape chars. Default : \\n, \\t, \\r
|
||||
|
||||
which_ones -- is a tuple of which escape chars we want to remove.
|
||||
|
|
@ -117,10 +118,10 @@ def replace_escape_chars(text, which_ones=('\n','\t','\r'), replace_by=u''):
|
|||
It defaults to '', so the escape chars are removed.
|
||||
"""
|
||||
for ec in which_ones:
|
||||
text = text.replace(ec, str_to_unicode(replace_by))
|
||||
return str_to_unicode(text)
|
||||
text = text.replace(ec, str_to_unicode(replace_by, encoding))
|
||||
return str_to_unicode(text, encoding)
|
||||
|
||||
def unquote_markup(text, keep=(), remove_illegal=True):
|
||||
def unquote_markup(text, keep=(), remove_illegal=True, encoding=None):
|
||||
"""
|
||||
This function receives markup as a text (always a unicode string or a utf-8 encoded string) and does the following:
|
||||
- removes entities (except the ones in 'keep') from any part of it that it's not inside a CDATA
|
||||
|
|
@ -138,7 +139,7 @@ def unquote_markup(text, keep=(), remove_illegal=True):
|
|||
offset = match_e
|
||||
yield txt[offset:]
|
||||
|
||||
text = str_to_unicode(text)
|
||||
text = str_to_unicode(text, encoding)
|
||||
ret_text = u''
|
||||
for fragment in _get_fragments(text, _cdata_re):
|
||||
if isinstance(fragment, basestring):
|
||||
|
|
|
|||
|
|
@ -64,13 +64,15 @@ def unique(list_, key=lambda x: x):
|
|||
return result
|
||||
|
||||
|
||||
def str_to_unicode(text, encoding='utf-8'):
|
||||
def str_to_unicode(text, encoding=None):
|
||||
"""Return the unicode representation of text in the given encoding. Unlike
|
||||
.encode(encoding) this function can be applied directly to a unicode
|
||||
object without the risk of double-decoding problems (which can happen if
|
||||
you don't use the default 'ascii' encoding)
|
||||
"""
|
||||
|
||||
if encoding is None:
|
||||
encoding = 'utf-8'
|
||||
if isinstance(text, str):
|
||||
return text.decode(encoding)
|
||||
elif isinstance(text, unicode):
|
||||
|
|
@ -78,13 +80,15 @@ def str_to_unicode(text, encoding='utf-8'):
|
|||
else:
|
||||
raise TypeError('str_to_unicode must receive a str or unicode object, got %s' % type(text).__name__)
|
||||
|
||||
def unicode_to_str(text, encoding='utf-8'):
|
||||
def unicode_to_str(text, encoding=None):
|
||||
"""Return the str representation of text in the given encoding. Unlike
|
||||
.encode(encoding) this function can be applied directly to a str
|
||||
object without the risk of double-decoding problems (which can happen if
|
||||
you don't use the default 'ascii' encoding)
|
||||
"""
|
||||
|
||||
if encoding is None:
|
||||
encoding = 'utf-8'
|
||||
if isinstance(text, unicode):
|
||||
return text.encode(encoding)
|
||||
elif isinstance(text, str):
|
||||
|
|
@ -216,3 +220,28 @@ def get_func_args(func):
|
|||
else:
|
||||
raise TypeError('%s is not callable' % type(func))
|
||||
return func_args
|
||||
|
||||
|
||||
def equal_attributes(obj1, obj2, attributes):
|
||||
"""Compare two objects attributes"""
|
||||
# not attributes given return False by default
|
||||
if not attributes:
|
||||
return False
|
||||
|
||||
for attr in attributes:
|
||||
# support callables like itemgetter
|
||||
if callable(attr):
|
||||
if not attr(obj1) == attr(obj2):
|
||||
return False
|
||||
else:
|
||||
# check that objects has attribute
|
||||
if not hasattr(obj1, attr):
|
||||
return False
|
||||
if not hasattr(obj2, attr):
|
||||
return False
|
||||
# compare object attributes
|
||||
if not getattr(obj1, attr) == getattr(obj2, attr):
|
||||
return False
|
||||
# all attributes equal
|
||||
return True
|
||||
|
||||
|
|
|
|||
|
|
@ -18,7 +18,8 @@ from scrapy.xlib.BeautifulSoup import BeautifulSoup
|
|||
from scrapy.http import Response, HtmlResponse
|
||||
|
||||
def body_or_str(obj, unicode=True):
|
||||
assert isinstance(obj, (Response, basestring)), "obj must be Response or basestring, not %s" % type(obj).__name__
|
||||
assert isinstance(obj, (Response, basestring)), \
|
||||
"obj must be Response or basestring, not %s" % type(obj).__name__
|
||||
if isinstance(obj, Response):
|
||||
return obj.body_as_unicode() if unicode else obj.body
|
||||
elif isinstance(obj, str):
|
||||
|
|
@ -26,16 +27,17 @@ def body_or_str(obj, unicode=True):
|
|||
else:
|
||||
return obj if unicode else obj.encode('utf-8')
|
||||
|
||||
BASEURL_RE = re.compile(r'<base\s+href\s*=\s*[\"\']\s*([^\"\'\s]+)\s*[\"\']', re.I)
|
||||
BASEURL_RE = re.compile(ur'<base\s+href\s*=\s*[\"\']\s*([^\"\'\s]+)\s*[\"\']', re.I)
|
||||
_baseurl_cache = weakref.WeakKeyDictionary()
|
||||
def get_base_url(response):
|
||||
""" Return the base url of the given response used to resolve relative links. """
|
||||
if response not in _baseurl_cache:
|
||||
match = BASEURL_RE.search(response.body[0:4096])
|
||||
_baseurl_cache[response] = match.group(1) if match else response.url
|
||||
match = BASEURL_RE.search(response.body_as_unicode()[0:4096])
|
||||
_baseurl_cache[response] = urljoin_rfc(response.url, match.group(1)) if match else response.url
|
||||
return _baseurl_cache[response]
|
||||
|
||||
META_REFRESH_RE = re.compile(ur'<meta[^>]*http-equiv[^>]*refresh[^>]*content\s*=\s*(?P<quote>["\'])(?P<int>\d+)\s*;\s*url=(?P<url>.*?)(?P=quote)', re.DOTALL | re.IGNORECASE)
|
||||
META_REFRESH_RE = re.compile(ur'<meta[^>]*http-equiv[^>]*refresh[^>]*content\s*=\s*(?P<quote>["\'])(?P<int>\d+)\s*;\s*url=(?P<url>.*?)(?P=quote)', \
|
||||
re.DOTALL | re.IGNORECASE)
|
||||
_metaref_cache = weakref.WeakKeyDictionary()
|
||||
def get_meta_refresh(response):
|
||||
"""Parse the http-equiv parameter of the HTML meta element from the given
|
||||
|
|
@ -46,9 +48,7 @@ def get_meta_refresh(response):
|
|||
If no meta redirect is found, (None, None) is returned.
|
||||
"""
|
||||
if response not in _metaref_cache:
|
||||
encoding = getattr(response, 'encoding', None) or 'utf-8'
|
||||
body_chunk = remove_entities(unicode(response.body[0:4096], encoding, \
|
||||
errors='ignore'))
|
||||
body_chunk = remove_entities(response.body_as_unicode()[0:4096])
|
||||
match = META_REFRESH_RE.search(body_chunk)
|
||||
if match:
|
||||
interval = int(match.group('int'))
|
||||
|
|
@ -92,7 +92,7 @@ def response_httprepr(response):
|
|||
s += response.body
|
||||
return s
|
||||
|
||||
def open_in_browser(response):
|
||||
def open_in_browser(response, _openfunc=webbrowser.open):
|
||||
"""Open the given response in a local web browser, populating the <base>
|
||||
tag for external links to work
|
||||
"""
|
||||
|
|
@ -106,4 +106,4 @@ def open_in_browser(response):
|
|||
fd, fname = tempfile.mkstemp('.html')
|
||||
os.write(fd, body)
|
||||
os.close(fd)
|
||||
webbrowser.open("file://%s" % fname)
|
||||
return _openfunc("file://%s" % fname)
|
||||
|
|
|
|||
|
|
@ -130,7 +130,8 @@ def add_or_replace_parameter(url, name, new_value, sep='&', url_is_quoted=False)
|
|||
name+'='+new_value)
|
||||
return next_url
|
||||
|
||||
def canonicalize_url(url, keep_blank_values=True, keep_fragments=False):
|
||||
def canonicalize_url(url, keep_blank_values=True, keep_fragments=False, \
|
||||
encoding=None):
|
||||
"""Canonicalize the given url by applying the following procedures:
|
||||
|
||||
- sort query arguments, first by key, then by value
|
||||
|
|
@ -147,7 +148,7 @@ def canonicalize_url(url, keep_blank_values=True, keep_fragments=False):
|
|||
For examples see the tests in scrapy.tests.test_utils_url
|
||||
"""
|
||||
|
||||
url = unicode_to_str(url)
|
||||
url = unicode_to_str(url, encoding)
|
||||
scheme, netloc, path, params, query, fragment = urlparse.urlparse(url)
|
||||
keyvals = cgi.parse_qsl(query, keep_blank_values)
|
||||
keyvals.sort()
|
||||
|
|
|
|||
Loading…
Reference in New Issue