Merge branch 'master' into process_spider_exception_generator
|
|
@ -12,8 +12,10 @@ dist
|
|||
.idea
|
||||
htmlcov/
|
||||
.coverage
|
||||
.pytest_cache/
|
||||
.coverage.*
|
||||
.cache/
|
||||
.pytest_cache/
|
||||
|
||||
# Windows
|
||||
Thumbs.db
|
||||
|
|
|
|||
|
|
@ -0,0 +1,24 @@
|
|||
platform: x86
|
||||
version: '{branch}-{build}'
|
||||
environment:
|
||||
matrix:
|
||||
- PYTHON: "C:\\Python36"
|
||||
TOX_ENV: py36
|
||||
|
||||
branches:
|
||||
only:
|
||||
- master
|
||||
- /d+\.\d+\.\d+[\w\-]*$/
|
||||
|
||||
install:
|
||||
- "SET PATH=%PYTHON%;%PYTHON%\\Scripts;%PATH%"
|
||||
- "SET TOX_TESTENV_PASSENV=HOME USERPROFILE HOMEPATH HOMEDRIVE"
|
||||
- "pip install -U tox"
|
||||
|
||||
build: false
|
||||
skip_tags: true
|
||||
test_script:
|
||||
- "tox -e %TOX_ENV%"
|
||||
|
||||
cache:
|
||||
- '%LOCALAPPDATA%\pip\cache'
|
||||
|
|
@ -45,7 +45,7 @@ master_doc = 'index'
|
|||
|
||||
# General information about the project.
|
||||
project = u'Scrapy'
|
||||
copyright = u'2008-2016, Scrapy developers'
|
||||
copyright = u'2008–2018, Scrapy developers'
|
||||
|
||||
# The version info for the project you're documenting, acts as replacement for
|
||||
# |version| and |release|, also used in various other places throughout the
|
||||
|
|
|
|||
|
|
@ -12,7 +12,7 @@ Contributing to Scrapy
|
|||
There are many ways to contribute to Scrapy. Here are some of them:
|
||||
|
||||
* Blog about Scrapy. Tell the world how you're using Scrapy. This will help
|
||||
newcomers with more examples and the Scrapy project to increase its
|
||||
newcomers with more examples and will help the Scrapy project to increase its
|
||||
visibility.
|
||||
|
||||
* Report bugs and request features in the `issue tracker`_, trying to follow
|
||||
|
|
@ -39,7 +39,7 @@ Reporting bugs
|
|||
trusted Scrapy developers, and its archives are not public.
|
||||
|
||||
Well-written bug reports are very helpful, so keep in mind the following
|
||||
guidelines when reporting a new bug.
|
||||
guidelines when you're going to report a new bug.
|
||||
|
||||
* check the :ref:`FAQ <faq>` first to see if your issue is addressed in a
|
||||
well-known question
|
||||
|
|
|
|||
|
|
@ -151,8 +151,7 @@ Solving specific problems
|
|||
topics/contracts
|
||||
topics/practices
|
||||
topics/broad-crawls
|
||||
topics/firefox
|
||||
topics/firebug
|
||||
topics/developer-tools
|
||||
topics/leaks
|
||||
topics/media-pipeline
|
||||
topics/deploy
|
||||
|
|
@ -175,11 +174,8 @@ Solving specific problems
|
|||
:doc:`topics/broad-crawls`
|
||||
Tune Scrapy for crawling a lot domains in parallel.
|
||||
|
||||
:doc:`topics/firefox`
|
||||
Learn how to scrape with Firefox and some useful add-ons.
|
||||
|
||||
:doc:`topics/firebug`
|
||||
Learn how to scrape efficiently using Firebug.
|
||||
:doc:`topics/developer-tools`
|
||||
Learn how to scrape with your browser's developer tools.
|
||||
|
||||
:doc:`topics/leaks`
|
||||
Learn how to find and get rid of memory leaks in your crawler.
|
||||
|
|
|
|||
|
|
@ -34,11 +34,11 @@ http://quotes.toscrape.com, following the pagination::
|
|||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {
|
||||
'text': quote.css('span.text::text').extract_first(),
|
||||
'author': quote.xpath('span/small/text()').extract_first(),
|
||||
'text': quote.css('span.text::text').get(),
|
||||
'author': quote.xpath('span/small/text()').get(),
|
||||
}
|
||||
|
||||
next_page = response.css('li.next a::attr("href")').extract_first()
|
||||
next_page = response.css('li.next a::attr("href")').get()
|
||||
if next_page is not None:
|
||||
yield response.follow(next_page, self.parse)
|
||||
|
||||
|
|
|
|||
|
|
@ -26,15 +26,26 @@ If you're already familiar with other languages, and want to learn Python
|
|||
quickly, we recommend reading through `Dive Into Python 3`_. Alternatively,
|
||||
you can follow the `Python Tutorial`_.
|
||||
|
||||
If you're new to programming and want to start with Python, you may find useful
|
||||
the online book `Learn Python The Hard Way`_. You can also take a look at `this
|
||||
list of Python resources for non-programmers`_.
|
||||
If you're new to programming and want to start with Python, the following books
|
||||
may be useful to you:
|
||||
|
||||
* `Automate the Boring Stuff With Python`_
|
||||
|
||||
* `How To Think Like a Computer Scientist`_
|
||||
|
||||
* `Learn Python 3 The Hard Way`_
|
||||
|
||||
You can also take a look at `this list of Python resources for non-programmers`_,
|
||||
as well as the `suggested resources in the learnpython-subreddit`_.
|
||||
|
||||
.. _Python: https://www.python.org/
|
||||
.. _this list of Python resources for non-programmers: https://wiki.python.org/moin/BeginnersGuide/NonProgrammers
|
||||
.. _Dive Into Python 3: http://www.diveintopython3.net
|
||||
.. _Python Tutorial: https://docs.python.org/3/tutorial
|
||||
.. _Learn Python The Hard Way: https://learnpythonthehardway.org/book/
|
||||
.. _Automate the Boring Stuff With Python: https://automatetheboringstuff.com/
|
||||
.. _How To Think Like a Computer Scientist: http://openbookproject.net/thinkcs/python/english3e/
|
||||
.. _Learn Python 3 The Hard Way: https://learnpythonthehardway.org/python3/
|
||||
.. _suggested resources in the learnpython-subreddit: https://www.reddit.com/r/learnpython/wiki/index#wiki_new_to_python.3F
|
||||
|
||||
|
||||
Creating a project
|
||||
|
|
@ -243,7 +254,7 @@ data.
|
|||
|
||||
To extract the text from the title above, you can do::
|
||||
|
||||
>>> response.css('title::text').extract()
|
||||
>>> response.css('title::text').getall()
|
||||
['Quotes to Scrape']
|
||||
|
||||
There are two things to note here: one is that we've added ``::text`` to the
|
||||
|
|
@ -251,32 +262,33 @@ CSS query, to mean we want to select only the text elements directly inside
|
|||
``<title>`` element. If we don't specify ``::text``, we'd get the full title
|
||||
element, including its tags::
|
||||
|
||||
>>> response.css('title').extract()
|
||||
>>> response.css('title').getall()
|
||||
['<title>Quotes to Scrape</title>']
|
||||
|
||||
The other thing is that the result of calling ``.extract()`` is a list, because
|
||||
we're dealing with an instance of :class:`~scrapy.selector.SelectorList`. When
|
||||
you know you just want the first result, as in this case, you can do::
|
||||
The other thing is that the result of calling ``.getall()`` is a list: it is
|
||||
possible that a selector returns more than one result, so we extract them all.
|
||||
When you know you just want the first result, as in this case, you can do::
|
||||
|
||||
>>> response.css('title::text').extract_first()
|
||||
>>> response.css('title::text').get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
As an alternative, you could've written::
|
||||
|
||||
>>> response.css('title::text')[0].extract()
|
||||
>>> response.css('title::text')[0].get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
However, using ``.extract_first()`` avoids an ``IndexError`` and returns
|
||||
``None`` when it doesn't find any element matching the selection.
|
||||
However, using ``.get()`` directly on a :class:`~scrapy.selector.SelectorList`
|
||||
instance avoids an ``IndexError`` and returns ``None`` when it doesn't
|
||||
find any element matching the selection.
|
||||
|
||||
There's a lesson here: for most scraping code, you want it to be resilient to
|
||||
errors due to things not being found on a page, so that even if some parts fail
|
||||
to be scraped, you can at least get **some** data.
|
||||
|
||||
Besides the :meth:`~scrapy.selector.Selector.extract` and
|
||||
:meth:`~scrapy.selector.SelectorList.extract_first` methods, you can also use
|
||||
the :meth:`~scrapy.selector.Selector.re` method to extract using `regular
|
||||
expressions`::
|
||||
Besides the :meth:`~scrapy.selector.SelectorList.getall` and
|
||||
:meth:`~scrapy.selector.SelectorList.get` methods, you can also use
|
||||
the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular
|
||||
expressions`_::
|
||||
|
||||
>>> response.css('title::text').re(r'Quotes.*')
|
||||
['Quotes to Scrape']
|
||||
|
|
@ -287,8 +299,8 @@ expressions`::
|
|||
|
||||
In order to find the proper CSS selectors to use, you might find useful opening
|
||||
the response page from the shell in your web browser using ``view(response)``.
|
||||
You can use your browser developer tools or extensions like Firebug (see
|
||||
sections about :ref:`topics-firebug` and :ref:`topics-firefox`).
|
||||
You can use your browser developer tools to inspect the HTML and come up
|
||||
with a selector (see section about :ref:`topics-developer-tools`).
|
||||
|
||||
`Selector Gadget`_ is also a nice tool to quickly find CSS selector for
|
||||
visually selected elements, which works in many browsers.
|
||||
|
|
@ -304,7 +316,7 @@ Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions::
|
|||
|
||||
>>> response.xpath('//title')
|
||||
[<Selector xpath='//title' data='<title>Quotes to Scrape</title>'>]
|
||||
>>> response.xpath('//title/text()').extract_first()
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
XPath expressions are very powerful, and are the foundation of Scrapy
|
||||
|
|
@ -373,17 +385,17 @@ variable, so that we can run our CSS selectors directly on a particular quote::
|
|||
Now, let's extract ``title``, ``author`` and the ``tags`` from that quote
|
||||
using the ``quote`` object we just created::
|
||||
|
||||
>>> title = quote.css("span.text::text").extract_first()
|
||||
>>> title = quote.css("span.text::text").get()
|
||||
>>> title
|
||||
'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'
|
||||
>>> author = quote.css("small.author::text").extract_first()
|
||||
>>> author = quote.css("small.author::text").get()
|
||||
>>> author
|
||||
'Albert Einstein'
|
||||
|
||||
Given that the tags are a list of strings, we can use the ``.extract()`` method
|
||||
Given that the tags are a list of strings, we can use the ``.getall()`` method
|
||||
to get all of them::
|
||||
|
||||
>>> tags = quote.css("div.tags a.tag::text").extract()
|
||||
>>> tags = quote.css("div.tags a.tag::text").getall()
|
||||
>>> tags
|
||||
['change', 'deep-thoughts', 'thinking', 'world']
|
||||
|
||||
|
|
@ -391,9 +403,9 @@ Having figured out how to extract each bit, we can now iterate over all the
|
|||
quotes elements and put them together into a Python dictionary::
|
||||
|
||||
>>> for quote in response.css("div.quote"):
|
||||
... text = quote.css("span.text::text").extract_first()
|
||||
... author = quote.css("small.author::text").extract_first()
|
||||
... tags = quote.css("div.tags a.tag::text").extract()
|
||||
... text = quote.css("span.text::text").get()
|
||||
... author = quote.css("small.author::text").get()
|
||||
... tags = quote.css("div.tags a.tag::text").getall()
|
||||
... print(dict(text=text, author=author, tags=tags))
|
||||
{'tags': ['change', 'deep-thoughts', 'thinking', 'world'], 'author': 'Albert Einstein', 'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'}
|
||||
{'tags': ['abilities', 'choices'], 'author': 'J.K. Rowling', 'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”'}
|
||||
|
|
@ -424,9 +436,9 @@ in the callback, as you can see below::
|
|||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {
|
||||
'text': quote.css('span.text::text').extract_first(),
|
||||
'author': quote.css('small.author::text').extract_first(),
|
||||
'tags': quote.css('div.tags a.tag::text').extract(),
|
||||
'text': quote.css('span.text::text').get(),
|
||||
'author': quote.css('small.author::text').get(),
|
||||
'tags': quote.css('div.tags a.tag::text').getall(),
|
||||
}
|
||||
|
||||
If you run this spider, it will output the extracted data with the log::
|
||||
|
|
@ -498,16 +510,22 @@ markup:
|
|||
|
||||
We can try extracting it in the shell::
|
||||
|
||||
>>> response.css('li.next a').extract_first()
|
||||
>>> response.css('li.next a').get()
|
||||
'<a href="/page/2/">Next <span aria-hidden="true">→</span></a>'
|
||||
|
||||
This gets the anchor element, but we want the attribute ``href``. For that,
|
||||
Scrapy supports a CSS extension that let's you select the attribute contents,
|
||||
like this::
|
||||
|
||||
>>> response.css('li.next a::attr(href)').extract_first()
|
||||
>>> response.css('li.next a::attr(href)').get()
|
||||
'/page/2/'
|
||||
|
||||
There is also an ``attrib`` property available
|
||||
(see :ref:`selecting-attributes` for more)::
|
||||
|
||||
>>> response.css('li.next a').attrib['href']
|
||||
'/page/2'
|
||||
|
||||
Let's see now our spider modified to recursively follow the link to the next
|
||||
page, extracting data from it::
|
||||
|
||||
|
|
@ -523,12 +541,12 @@ page, extracting data from it::
|
|||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {
|
||||
'text': quote.css('span.text::text').extract_first(),
|
||||
'author': quote.css('small.author::text').extract_first(),
|
||||
'tags': quote.css('div.tags a.tag::text').extract(),
|
||||
'text': quote.css('span.text::text').get(),
|
||||
'author': quote.css('small.author::text').get(),
|
||||
'tags': quote.css('div.tags a.tag::text').getall(),
|
||||
}
|
||||
|
||||
next_page = response.css('li.next a::attr(href)').extract_first()
|
||||
next_page = response.css('li.next a::attr(href)').get()
|
||||
if next_page is not None:
|
||||
next_page = response.urljoin(next_page)
|
||||
yield scrapy.Request(next_page, callback=self.parse)
|
||||
|
|
@ -574,12 +592,12 @@ As a shortcut for creating Request objects you can use
|
|||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {
|
||||
'text': quote.css('span.text::text').extract_first(),
|
||||
'author': quote.css('span small::text').extract_first(),
|
||||
'tags': quote.css('div.tags a.tag::text').extract(),
|
||||
'text': quote.css('span.text::text').get(),
|
||||
'author': quote.css('span small::text').get(),
|
||||
'tags': quote.css('div.tags a.tag::text').getall(),
|
||||
}
|
||||
|
||||
next_page = response.css('li.next a::attr(href)').extract_first()
|
||||
next_page = response.css('li.next a::attr(href)').get()
|
||||
if next_page is not None:
|
||||
yield response.follow(next_page, callback=self.parse)
|
||||
|
||||
|
|
@ -631,7 +649,7 @@ this time for scraping author information::
|
|||
|
||||
def parse_author(self, response):
|
||||
def extract_with_css(query):
|
||||
return response.css(query).extract_first().strip()
|
||||
return response.css(query).get(default='').strip()
|
||||
|
||||
yield {
|
||||
'name': extract_with_css('h3.author-title::text'),
|
||||
|
|
@ -700,11 +718,11 @@ with a specific tag, building the URL based on the argument::
|
|||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {
|
||||
'text': quote.css('span.text::text').extract_first(),
|
||||
'author': quote.css('small.author::text').extract_first(),
|
||||
'text': quote.css('span.text::text').get(),
|
||||
'author': quote.css('small.author::text').get(),
|
||||
}
|
||||
|
||||
next_page = response.css('li.next a::attr(href)').extract_first()
|
||||
next_page = response.css('li.next a::attr(href)').get()
|
||||
if next_page is not None:
|
||||
yield response.follow(next_page, self.parse)
|
||||
|
||||
|
|
@ -728,4 +746,3 @@ modeling the scraped data. If you prefer to play with an example project, check
|
|||
the :ref:`intro-examples` section.
|
||||
|
||||
.. _JSON: https://en.wikipedia.org/wiki/JSON
|
||||
.. _dirbot: https://github.com/scrapy/dirbot
|
||||
|
|
|
|||
|
Before Width: | Height: | Size: 43 KiB |
|
Before Width: | Height: | Size: 68 KiB |
|
Before Width: | Height: | Size: 88 KiB |
|
After Width: | Height: | Size: 53 KiB |
|
After Width: | Height: | Size: 10 KiB |
|
After Width: | Height: | Size: 81 KiB |
|
After Width: | Height: | Size: 44 KiB |
|
|
@ -458,9 +458,9 @@ Usage example::
|
|||
|
||||
>>> STATUS DEPTH LEVEL 1 <<<
|
||||
# Scraped Items ------------------------------------------------------------
|
||||
[{'name': u'Example item',
|
||||
'category': u'Furniture',
|
||||
'length': u'12 cm'}]
|
||||
[{'name': 'Example item',
|
||||
'category': 'Furniture',
|
||||
'length': '12 cm'}]
|
||||
|
||||
# Requests -----------------------------------------------------------------
|
||||
[]
|
||||
|
|
|
|||
|
|
@ -86,8 +86,11 @@ override three methods:
|
|||
.. method:: Contract.adjust_request_args(args)
|
||||
|
||||
This receives a ``dict`` as an argument containing default arguments
|
||||
for :class:`~scrapy.http.Request` object. Must return the same or a
|
||||
modified version of it.
|
||||
for request object. :class:`~scrapy.http.Request` is used by default,
|
||||
but this can be changed with the ``request_cls`` attribute.
|
||||
If multiple contracts in chain have this attribute defined, the last one is used.
|
||||
|
||||
Must return the same or a modified version of it.
|
||||
|
||||
.. method:: Contract.pre_process(response)
|
||||
|
||||
|
|
|
|||
|
|
@ -0,0 +1,268 @@
|
|||
.. _topics-developer-tools:
|
||||
|
||||
=================================================
|
||||
Using your browser's Developer Tools for scraping
|
||||
=================================================
|
||||
|
||||
Here is a general guide on how to use your browser's Developer Tools
|
||||
to ease the scraping process. Today almost all browsers come with
|
||||
built in `Developer Tools`_ and although we will use Firefox in this
|
||||
guide, the concepts are applicable to any other browser.
|
||||
|
||||
In this guide we'll introduce the basic tools to use from a browser's
|
||||
Developer Tools by scraping `quotes.toscrape.com`_.
|
||||
|
||||
.. _topics-livedom:
|
||||
|
||||
Caveats with inspecting the live browser DOM
|
||||
============================================
|
||||
|
||||
Since Developer Tools operate on a live browser DOM, what you'll actually see
|
||||
when inspecting the page source is not the original HTML, but a modified one
|
||||
after applying some browser clean up and executing Javascript code. Firefox,
|
||||
in particular, is known for adding ``<tbody>`` elements to tables. Scrapy, on
|
||||
the other hand, does not modify the original page HTML, so you won't be able to
|
||||
extract any data if you use ``<tbody>`` in your XPath expressions.
|
||||
|
||||
Therefore, you should keep in mind the following things:
|
||||
|
||||
* Disable Javascript while inspecting the DOM looking for XPaths to be
|
||||
used in Scrapy (in the Developer Tools settings click `Disable JavaScript`)
|
||||
|
||||
* Never use full XPath paths, use relative and clever ones based on attributes
|
||||
(such as ``id``, ``class``, ``width``, etc) or any identifying features like
|
||||
``contains(@href, 'image')``.
|
||||
|
||||
* Never include ``<tbody>`` elements in your XPath expressions unless you
|
||||
really know what you're doing
|
||||
|
||||
.. _topics-inspector:
|
||||
|
||||
Inspecting a website
|
||||
===================================
|
||||
|
||||
By far the most handy feature of the Developer Tools is the `Inspector`
|
||||
feature, which allows you to inspect the underlying HTML code of
|
||||
any webpage. To demonstrate the Inspector, let's look at the
|
||||
`quotes.toscrape.com`_-site.
|
||||
|
||||
On the site we have a total of ten quotes from various authors with specific
|
||||
tags, as well as the Top Ten Tags. Let's say we want to extract all the quotes
|
||||
on this page, without any meta-information about authors, tags, etc.
|
||||
|
||||
Instead of viewing the whole source code for the page, we can simply right click
|
||||
on a quote and select ``Inspect Element (Q)``, which opens up the `Inspector`.
|
||||
In it you should see something like this:
|
||||
|
||||
.. image:: _images/inspector_01.png
|
||||
:width: 777
|
||||
:height: 469
|
||||
:alt: Firefox's Inspector-tool
|
||||
|
||||
The interesting part for us is this:
|
||||
|
||||
.. code-block:: html
|
||||
|
||||
<div class="quote" itemscope="" itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">(...)</span>
|
||||
<span>(...)</span>
|
||||
<div class="tags">(...)</div>
|
||||
</div>
|
||||
|
||||
If you hover over the first ``div`` directly above the ``span`` tag highlighted
|
||||
in the screenshot, you'll see that the corresponding section of the webpage gets
|
||||
highlighted as well. So now we have a section, but we can't find our quote text
|
||||
anywhere.
|
||||
|
||||
The advantage of the `Inspector` is that it automatically expands and collapses
|
||||
sections and tags of a webpage, which greatly improves readability. You can
|
||||
expand and collapse a tag by clicking on the arrow in front of it or by double
|
||||
clicking directly on the tag. If we expand the ``span`` tag with the ``class=
|
||||
"text"`` we will see the quote-text we clicked on. The `Inspector` lets you
|
||||
copy XPaths to selected elements. Let's try it out: Right-click on the ``span``
|
||||
tag, select ``Copy > XPath`` and paste it in the scrapy shell like so::
|
||||
|
||||
$ scrapy shell "http://quotes.toscrape.com/"
|
||||
(...)
|
||||
>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall()
|
||||
['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”]
|
||||
|
||||
Adding ``text()`` at the end we are able to extract the first quote with this
|
||||
basic selector. But this XPath is not really that clever. All it does is
|
||||
go down a desired path in the source code starting from ``html``. So let's
|
||||
see if we can refine our XPath a bit:
|
||||
|
||||
If we check the `Inspector` again we'll see that directly beneath our
|
||||
expanded ``div`` tag we have nine identical ``div`` tags, each with the
|
||||
same attributes as our first. If we expand any of them, we'll see the same
|
||||
structure as with our first quote: Two ``span`` tags and one ``div`` tag. We can
|
||||
expand each ``span`` tag with the ``class="text"`` inside our ``div`` tags and
|
||||
see each quote:
|
||||
|
||||
.. code-block:: html
|
||||
|
||||
<div class="quote" itemscope="" itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">
|
||||
“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”
|
||||
</span>
|
||||
<span>(...)</span>
|
||||
<div class="tags">(...)</div>
|
||||
</div>
|
||||
|
||||
|
||||
With this knowledge we can refine our XPath: Instead of a path to follow,
|
||||
we'll simply select all ``span`` tags with the ``class="text"`` by using
|
||||
the `has-class-extension`_::
|
||||
|
||||
>>> response.xpath('//span[has-class("text")]/text()').getall()
|
||||
['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”,
|
||||
'“It is our choices, Harry, that show what we truly are, far more than our abilities.”',
|
||||
'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”',
|
||||
(...)]
|
||||
|
||||
And with one simple, cleverer XPath we are able to extract all quotes from
|
||||
the page. We could have constructed a loop over our first XPath to increase
|
||||
the number of the last ``div``, but this would have been unnecessarily
|
||||
complex and by simply constructing an XPath with ``has-class("text")``
|
||||
we were able to extract all quotes in one line.
|
||||
|
||||
The `Inspector` has a lot of other helpful features, such as searching in the
|
||||
source code or directly scrolling to an element you selected. Let's demonstrate
|
||||
a use case:
|
||||
|
||||
Say you want to find the ``Next`` button on the page. Type ``Next`` into the
|
||||
search bar on the top right of the `Inspector`. You should get two results.
|
||||
The first is a ``li`` tag with the ``class="text"``, the second the text
|
||||
of an ``a`` tag. Right click on the ``a`` tag and select ``Scroll into View``.
|
||||
If you hover over the tag, you'll see the button highlighted. From here
|
||||
we could easily create a :ref:`Link Extractor <topics-link-extractors>` to
|
||||
follow the pagination. On a simple site such as this, there may not be
|
||||
the need to find an element visually but the ``Scroll into View`` function
|
||||
can be quite useful on complex sites.
|
||||
|
||||
Note that the search bar can also be used to search for and test CSS
|
||||
selectors. For example, you could search for ``span.text`` to find
|
||||
all quote texts. Instead of a full text search, this searches for
|
||||
exactly the ``span`` tag with the ``class="text"`` in the page.
|
||||
|
||||
.. _topics-network-tool:
|
||||
|
||||
The Network-tool
|
||||
================
|
||||
While scraping you may come across dynamic webpages where some parts
|
||||
of the page are loaded dynamically through multiple requests. While
|
||||
this can be quite tricky, the `Network`-tool in the Developer Tools
|
||||
greatly facilitates this task. To demonstrate the Network-tool, let's
|
||||
take a look at the page `quotes.toscrape.com/scroll`_.
|
||||
|
||||
The page is quite similar to the basic `quotes.toscrape.com`_-page,
|
||||
but instead of the above-mentioned ``Next`` button, the page
|
||||
automatically loads new quotes when you scroll to the bottom. We
|
||||
could go ahead and try out different XPaths directly, but instead
|
||||
we'll check another quite useful command from the scrapy shell::
|
||||
|
||||
$ scrapy shell "quotes.toscrape.com/scroll"
|
||||
(...)
|
||||
>>> view(response)
|
||||
|
||||
A browser window should open with the webpage but with one
|
||||
crucial difference: Instead of the quotes we just see a greenish
|
||||
bar with the word ``Loading...``.
|
||||
|
||||
.. image:: _images/network_01.png
|
||||
:width: 777
|
||||
:height: 296
|
||||
:alt: Response from quotes.toscrape.com/scroll
|
||||
|
||||
The ``view(response)`` command let's us view the response our
|
||||
shell or later our spider receives from the server. Here we see
|
||||
that some basic template is loaded which includes the title,
|
||||
the login-button and the footer, but the quotes are missing. This
|
||||
tells us that the quotes are being loaded from a different request
|
||||
than ``quotes.toscrape/scroll``.
|
||||
|
||||
If you click on the ``Network`` tab, you will probably only see
|
||||
two entries. The first thing we do is enable persistent logs by
|
||||
clicking on ``Persist Logs``. If this option is disabled, the
|
||||
log is automatically cleared each time you navigate to a different
|
||||
page. Enabling this option is a good default, since it gives us
|
||||
control on when to clear the logs.
|
||||
|
||||
If we reload the page now, you'll see the log get populated with six
|
||||
new requests.
|
||||
|
||||
.. image:: _images/network_02.png
|
||||
:width: 777
|
||||
:height: 241
|
||||
:alt: Network tab with persistent logs and requests
|
||||
|
||||
Here we see every request that has been made when reloading the page
|
||||
and can inspect each request and its response. So let's find out
|
||||
where our quotes are coming from:
|
||||
|
||||
First click on the request with the name ``scroll``. On the right
|
||||
you can now inspect the request. In ``Headers`` you'll find details
|
||||
about the request headers, such as the URL, the method, the IP-address,
|
||||
and so on. We'll ignore the other tabs and click directly on ``Reponse``.
|
||||
|
||||
What you should see in the ``Preview`` pane is the rendered HTML-code,
|
||||
that is exactly what we saw when we called ``view(response)`` in the
|
||||
shell. Accordingly the ``type`` of the request in the log is ``html``.
|
||||
The other requests have types like ``css`` or ``js``, but what
|
||||
interests us is the one request called ``quotes?page=1`` with the
|
||||
type ``json``.
|
||||
|
||||
If we click on this request, we see that the request URL is
|
||||
``http://quotes.toscrape.com/api/quotes?page=1`` and the response
|
||||
is a JSON-object that contains our quotes. We can also right-click
|
||||
on the request and open ``Open in new tab`` to get a better overview.
|
||||
|
||||
.. image:: _images/network_03.png
|
||||
:width: 777
|
||||
:height: 375
|
||||
:alt: JSON-object returned from the quotes.toscrape API
|
||||
|
||||
With this response we can now easily parse the JSON-object and
|
||||
also request each page to get every quote on the site::
|
||||
|
||||
import scrapy
|
||||
import json
|
||||
|
||||
|
||||
class QuoteSpider(scrapy.Spider):
|
||||
name = 'quote'
|
||||
allowed_domains = ['quotes.toscrape.com']
|
||||
page = 1
|
||||
start_urls = ['http://quotes.toscrape.com/api/quotes?page=1]
|
||||
|
||||
def parse(self, response):
|
||||
data = json.loads(response.text)
|
||||
for quote in data["quotes"]:
|
||||
yield {"quote": quote["text"]}
|
||||
if data["has_next"]:
|
||||
self.page += 1
|
||||
url = "http://quotes.toscrape.com/api/quotes?page={}".format(self.page)
|
||||
yield scrapy.Request(url=url, callback=self.parse)
|
||||
|
||||
This spider starts at the first page of the quotes-API. With each
|
||||
response, we parse the ``response.text`` and assign it to ``data``.
|
||||
This lets us operate on the JSON-object like on a Python dictionary.
|
||||
We iterate through the ``quotes`` and print out the ``quote["text"]``.
|
||||
If the handy ``has_next`` element is ``true`` (try loading
|
||||
`quotes.toscrape.com/api/quotes?page=10`_ in your browser or a
|
||||
page-number greater than 10), we increment the ``page`` attribute
|
||||
and ``yield`` a new request, inserting the incremented page-number
|
||||
into our ``url``.
|
||||
|
||||
You can see that with a few inspections in the `Network`-tool we
|
||||
were able to easily replicate the dynamic requests of the scrolling
|
||||
functionality of the page. Crawling dynamic pages can be quite
|
||||
daunting and pages can be very complex, but it (mostly) boils down
|
||||
to identifying the correct request and replicating it in your spider.
|
||||
|
||||
.. _Developer Tools: https://en.wikipedia.org/wiki/Web_development_tools
|
||||
.. _quotes.toscrape.com: http://quotes.toscrape.com
|
||||
.. _quotes.toscrape.com/scroll: quotes.toscrape.com/scroll/
|
||||
.. _quotes.toscrape.com/api/quotes?page=10: http://quotes.toscrape.com/api/quotes?page=10
|
||||
.. _has-class-extension: https://parsel.readthedocs.io/en/latest/usage.html#other-xpath-extensions
|
||||
|
||||
|
|
@ -1,167 +0,0 @@
|
|||
.. _topics-firebug:
|
||||
|
||||
==========================
|
||||
Using Firebug for scraping
|
||||
==========================
|
||||
|
||||
.. note:: Google Directory, the example website used in this guide is no longer
|
||||
available as it `has been shut down by Google`_. The concepts in this guide
|
||||
are still valid though. If you want to update this guide to use a new
|
||||
(working) site, your contribution will be more than welcome!. See :ref:`topics-contributing`
|
||||
for information on how to do so.
|
||||
|
||||
Introduction
|
||||
============
|
||||
|
||||
This document explains how to use `Firebug`_ (a Firefox add-on) to make the
|
||||
scraping process easier and more fun. For other useful Firefox add-ons see
|
||||
:ref:`topics-firefox-addons`. There are some caveats with using Firefox add-ons
|
||||
to inspect pages, see :ref:`topics-firefox-livedom`.
|
||||
|
||||
In this example, we'll show how to use `Firebug`_ to scrape data from the
|
||||
`Google Directory`_, which contains the same data as the `Open Directory
|
||||
Project`_ used in the :ref:`tutorial <intro-tutorial>` but with a different
|
||||
face.
|
||||
|
||||
.. _Firebug: https://getfirebug.com/
|
||||
.. _Google Directory: http://directory.google.com/
|
||||
.. _Open Directory Project: http://www.dmoz.org
|
||||
|
||||
Firebug comes with a very useful feature called `Inspect Element`_ which allows
|
||||
you to inspect the HTML code of the different page elements just by hovering
|
||||
your mouse over them. Otherwise you would have to search for the tags manually
|
||||
through the HTML body which can be a very tedious task.
|
||||
|
||||
.. _Inspect Element: https://www.youtube.com/watch?v=-pT_pDe54aA
|
||||
|
||||
In the following screenshot you can see the `Inspect Element`_ tool in action.
|
||||
|
||||
.. image:: _images/firebug1.png
|
||||
:width: 913
|
||||
:height: 600
|
||||
:alt: Inspecting elements with Firebug
|
||||
|
||||
At first sight, we can see that the directory is divided in categories, which
|
||||
are also divided in subcategories.
|
||||
|
||||
However, it seems that there are more subcategories than the ones being shown
|
||||
in this page, so we'll keep looking:
|
||||
|
||||
.. image:: _images/firebug2.png
|
||||
:width: 819
|
||||
:height: 629
|
||||
:alt: Inspecting elements with Firebug
|
||||
|
||||
As expected, the subcategories contain links to other subcategories, and also
|
||||
links to actual websites, which is the purpose of the directory.
|
||||
|
||||
Getting links to follow
|
||||
=======================
|
||||
|
||||
By looking at the category URLs we can see they share a pattern:
|
||||
|
||||
http://directory.google.com/Category/Subcategory/Another_Subcategory
|
||||
|
||||
Once we know that, we are able to construct a regular expression to follow
|
||||
those links. For example, the following one::
|
||||
|
||||
directory\.google\.com/[A-Z][a-zA-Z_/]+$
|
||||
|
||||
So, based on that regular expression we can create the first crawling rule::
|
||||
|
||||
Rule(LinkExtractor(allow='directory.google.com/[A-Z][a-zA-Z_/]+$', ),
|
||||
'parse_category',
|
||||
follow=True,
|
||||
),
|
||||
|
||||
The :class:`~scrapy.spiders.Rule` object instructs
|
||||
:class:`~scrapy.spiders.CrawlSpider` based spiders how to follow the
|
||||
category links. ``parse_category`` will be a method of the spider which will
|
||||
process and extract data from those pages.
|
||||
|
||||
This is how the spider would look so far::
|
||||
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
from scrapy.spiders import CrawlSpider, Rule
|
||||
|
||||
class GoogleDirectorySpider(CrawlSpider):
|
||||
name = 'directory.google.com'
|
||||
allowed_domains = ['directory.google.com']
|
||||
start_urls = ['http://directory.google.com/']
|
||||
|
||||
rules = (
|
||||
Rule(LinkExtractor(allow='directory\.google\.com/[A-Z][a-zA-Z_/]+$'),
|
||||
'parse_category', follow=True,
|
||||
),
|
||||
)
|
||||
|
||||
def parse_category(self, response):
|
||||
# write the category page data extraction code here
|
||||
pass
|
||||
|
||||
|
||||
Extracting the data
|
||||
===================
|
||||
|
||||
Now we're going to write the code to extract data from those pages.
|
||||
|
||||
With the help of Firebug, we'll take a look at some page containing links to
|
||||
websites (say http://directory.google.com/Top/Arts/Awards/) and find out how we can
|
||||
extract those links using :ref:`Selectors <topics-selectors>`. We'll also
|
||||
use the :ref:`Scrapy shell <topics-shell>` to test those XPath's and make sure
|
||||
they work as we expect.
|
||||
|
||||
.. image:: _images/firebug3.png
|
||||
:width: 965
|
||||
:height: 751
|
||||
:alt: Inspecting elements with Firebug
|
||||
|
||||
As you can see, the page markup is not very descriptive: the elements don't
|
||||
contain ``id``, ``class`` or any attribute that clearly identifies them, so
|
||||
we'll use the ranking bars as a reference point to select the data to extract
|
||||
when we construct our XPaths.
|
||||
|
||||
After using FireBug, we can see that each link is inside a ``td`` tag, which is
|
||||
itself inside a ``tr`` tag that also contains the link's ranking bar (in
|
||||
another ``td``).
|
||||
|
||||
So we can select the ranking bar, then find its parent (the ``tr``), and then
|
||||
finally, the link's ``td`` (which contains the data we want to scrape).
|
||||
|
||||
This results in the following XPath::
|
||||
|
||||
//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td//a
|
||||
|
||||
It's important to use the :ref:`Scrapy shell <topics-shell>` to test these
|
||||
complex XPath expressions and make sure they work as expected.
|
||||
|
||||
Basically, that expression will look for the ranking bar's ``td`` element, and
|
||||
then select any ``td`` element who has a descendant ``a`` element whose
|
||||
``href`` attribute contains the string ``#pagerank``"
|
||||
|
||||
Of course, this is not the only XPath, and maybe not the simpler one to select
|
||||
that data. Another approach could be, for example, to find any ``font`` tags
|
||||
that have that grey colour of the links,
|
||||
|
||||
Finally, we can write our ``parse_category()`` method::
|
||||
|
||||
def parse_category(self, response):
|
||||
# The path to website links in directory page
|
||||
links = response.xpath('//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td/font')
|
||||
|
||||
for link in links:
|
||||
item = DirectoryItem()
|
||||
item['name'] = link.xpath('a/text()').extract()
|
||||
item['url'] = link.xpath('a/@href').extract()
|
||||
item['description'] = link.xpath('font[2]/text()').extract()
|
||||
yield item
|
||||
|
||||
|
||||
Be aware that you may find some elements which appear in Firebug but
|
||||
not in the original HTML, such as the typical case of ``<tbody>``
|
||||
elements.
|
||||
|
||||
or tags which Therefer in page HTML
|
||||
sources may on Firebug inspects the live DOM
|
||||
|
||||
.. _has been shut down by Google: https://searchenginewatch.com/sew/news/2096661/google-directory-shut
|
||||
|
|
@ -1,82 +0,0 @@
|
|||
.. _topics-firefox:
|
||||
|
||||
==========================
|
||||
Using Firefox for scraping
|
||||
==========================
|
||||
|
||||
Here is a list of tips and advice on using Firefox for scraping, along with a
|
||||
list of useful Firefox add-ons to ease the scraping process.
|
||||
|
||||
.. _topics-firefox-livedom:
|
||||
|
||||
Caveats with inspecting the live browser DOM
|
||||
============================================
|
||||
|
||||
Since Firefox add-ons operate on a live browser DOM, what you'll actually see
|
||||
when inspecting the page source is not the original HTML, but a modified one
|
||||
after applying some browser clean up and executing Javascript code. Firefox,
|
||||
in particular, is known for adding ``<tbody>`` elements to tables. Scrapy, on
|
||||
the other hand, does not modify the original page HTML, so you won't be able to
|
||||
extract any data if you use ``<tbody>`` in your XPath expressions.
|
||||
|
||||
Therefore, you should keep in mind the following things when working with
|
||||
Firefox and XPath:
|
||||
|
||||
* Disable Firefox Javascript while inspecting the DOM looking for XPaths to be
|
||||
used in Scrapy
|
||||
|
||||
* Never use full XPath paths, use relative and clever ones based on attributes
|
||||
(such as ``id``, ``class``, ``width``, etc) or any identifying features like
|
||||
``contains(@href, 'image')``.
|
||||
|
||||
* Never include ``<tbody>`` elements in your XPath expressions unless you
|
||||
really know what you're doing
|
||||
|
||||
.. _topics-firefox-addons:
|
||||
|
||||
Useful Firefox add-ons for scraping
|
||||
===================================
|
||||
|
||||
Firebug
|
||||
-------
|
||||
|
||||
`Firebug`_ is a widely known tool among web developers and it's also very
|
||||
useful for scraping. In particular, its `Inspect Element`_ feature comes very
|
||||
handy when you need to construct the XPaths for extracting data because it
|
||||
allows you to view the HTML code of each page element while moving your mouse
|
||||
over it.
|
||||
|
||||
See :ref:`topics-firebug` for a detailed guide on how to use Firebug with
|
||||
Scrapy.
|
||||
|
||||
XPather
|
||||
-------
|
||||
|
||||
`XPather`_ allows you to test XPath expressions directly on the pages.
|
||||
|
||||
XPath Checker
|
||||
-------------
|
||||
|
||||
`XPath Checker`_ is another Firefox add-on for testing XPaths on your pages.
|
||||
|
||||
Tamper Data
|
||||
-----------
|
||||
|
||||
`Tamper Data`_ is a Firefox add-on which allows you to view and modify the HTTP
|
||||
request headers sent by Firefox. Firebug also allows to view HTTP headers, but
|
||||
not to modify them.
|
||||
|
||||
Firecookie
|
||||
----------
|
||||
|
||||
`Firecookie`_ makes it easier to view and manage cookies. You can use this
|
||||
extension to create a new cookie, delete existing cookies, see a list of cookies
|
||||
for the current site, manage cookies permissions and a lot more.
|
||||
|
||||
.. _Firebug: https://getfirebug.com/
|
||||
.. _Inspect Element: https://www.youtube.com/watch?v=-pT_pDe54aA
|
||||
.. _XPather: https://addons.mozilla.org/en-US/firefox/addon/xpather/
|
||||
.. _XPath Checker: https://addons.mozilla.org/en-US/firefox/addon/xpath-checker/
|
||||
.. _Tamper Data: https://addons.mozilla.org/en-US/firefox/addon/tamper-data/
|
||||
.. _Firecookie: https://addons.mozilla.org/en-US/firefox/addon/firecookie/
|
||||
|
||||
|
|
@ -86,7 +86,7 @@ Creating items
|
|||
::
|
||||
|
||||
>>> product = Product(name='Desktop PC', price=1000)
|
||||
>>> print product
|
||||
>>> print(product)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
Getting field values
|
||||
|
|
@ -161,11 +161,11 @@ Other common tasks
|
|||
Copying items::
|
||||
|
||||
>>> product2 = Product(product)
|
||||
>>> print product2
|
||||
>>> print(product2)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
>>> product3 = product2.copy()
|
||||
>>> print product3
|
||||
>>> print(product3)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
Creating dicts from items::
|
||||
|
|
|
|||
|
|
@ -84,7 +84,7 @@ So, for example, this won't work::
|
|||
return scrapy.Request('http://www.example.com', callback=lambda r: self.other_callback(r, somearg))
|
||||
|
||||
def other_callback(self, response, somearg):
|
||||
print "the argument passed is:", somearg
|
||||
print("the argument passed is: %s" % somearg)
|
||||
|
||||
But this will::
|
||||
|
||||
|
|
@ -94,7 +94,7 @@ But this will::
|
|||
|
||||
def other_callback(self, response):
|
||||
somearg = response.meta['somearg']
|
||||
print "the argument passed is:", somearg
|
||||
print("the argument passed is: %s" % somearg)
|
||||
|
||||
If you wish to log the requests that couldn't be serialized, you can set the
|
||||
:setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page.
|
||||
|
|
|
|||
|
|
@ -678,10 +678,10 @@ Here is a list of all built-in processors:
|
|||
>>> from scrapy.loader.processors import Join
|
||||
>>> proc = Join()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
u'one two three'
|
||||
'one two three'
|
||||
>>> proc = Join('<br>')
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
u'one<br>two<br>three'
|
||||
'one<br>two<br>three'
|
||||
|
||||
.. class:: Compose(\*functions, \**default_loader_context)
|
||||
|
||||
|
|
@ -744,9 +744,9 @@ Here is a list of all built-in processors:
|
|||
... return None if x == 'world' else x
|
||||
...
|
||||
>>> from scrapy.loader.processors import MapCompose
|
||||
>>> proc = MapCompose(filter_world, unicode.upper)
|
||||
>>> proc([u'hello', u'world', u'this', u'is', u'scrapy'])
|
||||
[u'HELLO, u'THIS', u'IS', u'SCRAPY']
|
||||
>>> proc = MapCompose(filter_world, str.upper)
|
||||
>>> proc(['hello', 'world', 'this', 'is', 'scrapy'])
|
||||
['HELLO, 'THIS', 'IS', 'SCRAPY']
|
||||
|
||||
As with the Compose processor, functions can receive Loader contexts, and
|
||||
constructor keyword arguments are used as default context values. See
|
||||
|
|
@ -772,7 +772,7 @@ Here is a list of all built-in processors:
|
|||
>>> import json
|
||||
>>> proc_single_json_str = Compose(json.loads, SelectJmes("foo"))
|
||||
>>> proc_single_json_str('{"foo": "bar"}')
|
||||
u'bar'
|
||||
'bar'
|
||||
>>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo')))
|
||||
>>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]')
|
||||
[u'bar']
|
||||
['bar']
|
||||
|
|
|
|||
|
|
@ -871,7 +871,7 @@ LOG_STDOUT
|
|||
Default: ``False``
|
||||
|
||||
If ``True``, all standard output (and error) of your process will be redirected
|
||||
to the log. For example if you ``print 'hello'`` it will appear in the Scrapy
|
||||
to the log. For example if you ``print('hello')`` it will appear in the Scrapy
|
||||
log.
|
||||
|
||||
.. setting:: LOG_SHORT_NAMES
|
||||
|
|
|
|||
|
|
@ -179,13 +179,13 @@ all start with the ``[s]`` prefix)::
|
|||
|
||||
After that, we can start playing with the objects::
|
||||
|
||||
>>> response.xpath('//title/text()').extract_first()
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
|
||||
|
||||
>>> fetch("https://reddit.com")
|
||||
|
||||
>>> response.xpath('//title/text()').extract()
|
||||
['reddit: the front page of the internet']
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'reddit: the front page of the internet'
|
||||
|
||||
>>> request = request.replace(method="POST")
|
||||
|
||||
|
|
|
|||
|
|
@ -279,6 +279,22 @@ request_dropped
|
|||
:param spider: the spider that yielded the request
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
request_reached_downloader
|
||||
---------------------------
|
||||
|
||||
.. signal:: request_reached_downloader
|
||||
.. function:: request_reached_downloader(request, spider)
|
||||
|
||||
Sent when a :class:`~scrapy.http.Request` reached downloader.
|
||||
|
||||
The signal does not support returning deferreds from their handlers.
|
||||
|
||||
:param request: the request that reached downloader
|
||||
:type request: :class:`~scrapy.http.Request` object
|
||||
|
||||
:param spider: the spider that yielded the request
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
response_received
|
||||
-----------------
|
||||
|
||||
|
|
|
|||
|
|
@ -229,11 +229,11 @@ Return multiple Requests and items from a single callback::
|
|||
]
|
||||
|
||||
def parse(self, response):
|
||||
for h3 in response.xpath('//h3').extract():
|
||||
for h3 in response.xpath('//h3').getall():
|
||||
yield {"title": h3}
|
||||
|
||||
for url in response.xpath('//a/@href').extract():
|
||||
yield scrapy.Request(url, callback=self.parse)
|
||||
for href in response.xpath('//a/@href').getall():
|
||||
yield scrapy.Request(response.urljoin(href), self.parse)
|
||||
|
||||
Instead of :attr:`~.start_urls` you can use :meth:`~.start_requests` directly;
|
||||
to give data more structure you can use :ref:`topics-items`::
|
||||
|
|
@ -251,11 +251,11 @@ to give data more structure you can use :ref:`topics-items`::
|
|||
yield scrapy.Request('http://www.example.com/3.html', self.parse)
|
||||
|
||||
def parse(self, response):
|
||||
for h3 in response.xpath('//h3').extract():
|
||||
for h3 in response.xpath('//h3').getall():
|
||||
yield MyItem(title=h3)
|
||||
|
||||
for url in response.xpath('//a/@href').extract():
|
||||
yield scrapy.Request(url, callback=self.parse)
|
||||
for href in response.xpath('//a/@href').getall():
|
||||
yield scrapy.Request(response.urljoin(href), self.parse)
|
||||
|
||||
.. _spiderargs:
|
||||
|
||||
|
|
@ -434,8 +434,8 @@ Let's now take a look at an example CrawlSpider with rules::
|
|||
self.logger.info('Hi, this is an item page! %s', response.url)
|
||||
item = scrapy.Item()
|
||||
item['id'] = response.xpath('//td[@id="item_id"]/text()').re(r'ID: (\d+)')
|
||||
item['name'] = response.xpath('//td[@id="item_name"]/text()').extract()
|
||||
item['description'] = response.xpath('//td[@id="item_description"]/text()').extract()
|
||||
item['name'] = response.xpath('//td[@id="item_name"]/text()').get()
|
||||
item['description'] = response.xpath('//td[@id="item_description"]/text()').get()
|
||||
return item
|
||||
|
||||
|
||||
|
|
@ -545,12 +545,12 @@ These spiders are pretty easy to use, let's have a look at one example::
|
|||
itertag = 'item'
|
||||
|
||||
def parse_node(self, response, node):
|
||||
self.logger.info('Hi, this is a <%s> node!: %s', self.itertag, ''.join(node.extract()))
|
||||
self.logger.info('Hi, this is a <%s> node!: %s', self.itertag, ''.join(node.getall()))
|
||||
|
||||
item = TestItem()
|
||||
item['id'] = node.xpath('@id').extract()
|
||||
item['name'] = node.xpath('name').extract()
|
||||
item['description'] = node.xpath('description').extract()
|
||||
item['id'] = node.xpath('@id').get()
|
||||
item['name'] = node.xpath('name').get()
|
||||
item['description'] = node.xpath('description').get()
|
||||
return item
|
||||
|
||||
Basically what we did up there was to create a spider that downloads a feed from
|
||||
|
|
|
|||
|
|
@ -6,5 +6,5 @@ queuelib
|
|||
w3lib>=1.17.0
|
||||
six>=1.5.2
|
||||
PyDispatcher>=2.0.5
|
||||
parsel>=1.4
|
||||
parsel>=1.5
|
||||
service_identity
|
||||
|
|
|
|||
|
|
@ -6,5 +6,5 @@ queuelib>=1.1.1
|
|||
w3lib>=1.17.0
|
||||
six>=1.5.2
|
||||
PyDispatcher>=2.0.5
|
||||
parsel>=1.4
|
||||
parsel>=1.5
|
||||
service_identity
|
||||
|
|
|
|||
|
|
@ -107,8 +107,8 @@ class Command(ScrapyCommand):
|
|||
string.Template(path).substitute(project_name=project_name))
|
||||
render_templatefile(tplfile, project_name=project_name,
|
||||
ProjectName=string_camelcase(project_name))
|
||||
print("New Scrapy project %r, using template directory %r, created in:" % \
|
||||
(project_name, self.templates_dir))
|
||||
print("New Scrapy project '%s', using template directory '%s', "
|
||||
"created in:" % (project_name, self.templates_dir))
|
||||
print(" %s\n" % abspath(project_dir))
|
||||
print("You can start your first spider with:")
|
||||
print(" cd %s" % project_dir)
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import sys
|
||||
import re
|
||||
from functools import wraps
|
||||
from inspect import getmembers
|
||||
from unittest import TestCase
|
||||
|
||||
from scrapy.http import Request
|
||||
|
|
@ -17,7 +18,7 @@ class ContractsManager(object):
|
|||
|
||||
def tested_methods_from_spidercls(self, spidercls):
|
||||
methods = []
|
||||
for key, value in vars(spidercls).items():
|
||||
for key, value in getmembers(spidercls):
|
||||
if (callable(value) and value.__doc__ and
|
||||
re.search(r'^\s*@', value.__doc__, re.MULTILINE)):
|
||||
methods.append(key)
|
||||
|
|
@ -41,23 +42,38 @@ class ContractsManager(object):
|
|||
requests = []
|
||||
for method in self.tested_methods_from_spidercls(type(spider)):
|
||||
bound_method = spider.__getattribute__(method)
|
||||
requests.append(self.from_method(bound_method, results))
|
||||
try:
|
||||
requests.append(self.from_method(bound_method, results))
|
||||
except Exception:
|
||||
case = _create_testcase(bound_method, 'contract')
|
||||
results.addError(case, sys.exc_info())
|
||||
|
||||
return requests
|
||||
|
||||
def from_method(self, method, results):
|
||||
contracts = self.extract_contracts(method)
|
||||
if contracts:
|
||||
request_cls = Request
|
||||
for contract in contracts:
|
||||
if contract.request_cls is not None:
|
||||
request_cls = contract.request_cls
|
||||
|
||||
# calculate request args
|
||||
args, kwargs = get_spec(Request.__init__)
|
||||
args, kwargs = get_spec(request_cls.__init__)
|
||||
|
||||
# Don't filter requests to allow
|
||||
# testing different callbacks on the same URL.
|
||||
kwargs['dont_filter'] = True
|
||||
kwargs['callback'] = method
|
||||
|
||||
for contract in contracts:
|
||||
kwargs = contract.adjust_request_args(kwargs)
|
||||
|
||||
# create and prepare request
|
||||
args.remove('self')
|
||||
|
||||
# check if all positional arguments are defined in kwargs
|
||||
if set(args).issubset(set(kwargs)):
|
||||
request = Request(**kwargs)
|
||||
request = request_cls(**kwargs)
|
||||
|
||||
# execute pre and post hooks in order
|
||||
for contract in reversed(contracts):
|
||||
|
|
@ -84,7 +100,7 @@ class ContractsManager(object):
|
|||
|
||||
def eb_wrapper(failure):
|
||||
case = _create_testcase(method, 'errback')
|
||||
exc_info = failure.value, failure.type, failure.getTracebackObject()
|
||||
exc_info = failure.type, failure.value, failure.getTracebackObject()
|
||||
results.addError(case, exc_info)
|
||||
|
||||
request.callback = cb_wrapper
|
||||
|
|
@ -93,6 +109,7 @@ class ContractsManager(object):
|
|||
|
||||
class Contract(object):
|
||||
""" Abstract class for contracts """
|
||||
request_cls = None
|
||||
|
||||
def __init__(self, method, *args):
|
||||
self.testcase_pre = _create_testcase(method, '@%s pre-hook' % self.name)
|
||||
|
|
|
|||
|
|
@ -129,6 +129,9 @@ class Downloader(object):
|
|||
return response
|
||||
|
||||
slot.active.add(request)
|
||||
self.signals.send_catch_log(signal=signals.request_reached_downloader,
|
||||
request=request,
|
||||
spider=spider)
|
||||
deferred = defer.Deferred().addBoth(_deactivate)
|
||||
slot.queue.append((request, deferred))
|
||||
self._process_queue(spider, slot)
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ import logging
|
|||
from os.path import join, exists
|
||||
|
||||
from scrapy.utils.reqser import request_to_dict, request_from_dict
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.misc import load_object, create_instance
|
||||
from scrapy.utils.job import job_dir
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
@ -26,7 +26,7 @@ class Scheduler(object):
|
|||
def from_crawler(cls, crawler):
|
||||
settings = crawler.settings
|
||||
dupefilter_cls = load_object(settings['DUPEFILTER_CLASS'])
|
||||
dupefilter = dupefilter_cls.from_settings(settings)
|
||||
dupefilter = create_instance(dupefilter_cls, settings, crawler)
|
||||
pqclass = load_object(settings['SCHEDULER_PRIORITY_QUEUE'])
|
||||
dqclass = load_object(settings['SCHEDULER_DISK_QUEUE'])
|
||||
mqclass = load_object(settings['SCHEDULER_MEMORY_QUEUE'])
|
||||
|
|
|
|||
|
|
@ -21,7 +21,7 @@ from w3lib.url import file_uri_to_path
|
|||
from scrapy import signals
|
||||
from scrapy.utils.ftp import ftp_makedirs_cwd
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.python import without_none_values
|
||||
from scrapy.utils.boto import is_botocore
|
||||
|
|
@ -93,12 +93,29 @@ class FileFeedStorage(object):
|
|||
|
||||
class S3FeedStorage(BlockingFeedStorage):
|
||||
|
||||
def __init__(self, uri):
|
||||
from scrapy.conf import settings
|
||||
def __init__(self, uri, access_key=None, secret_key=None):
|
||||
# BEGIN Backwards compatibility for initialising without keys (and
|
||||
# without using from_crawler)
|
||||
no_defaults = access_key is None and secret_key is None
|
||||
if no_defaults:
|
||||
from scrapy.conf import settings
|
||||
if 'AWS_ACCESS_KEY_ID' in settings or 'AWS_SECRET_ACCESS_KEY' in settings:
|
||||
import warnings
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
warnings.warn(
|
||||
"Initialising `scrapy.extensions.feedexport.S3FeedStorage` "
|
||||
"without AWS keys is deprecated. Please supply credentials or "
|
||||
"use the `from_crawler()` constructor.",
|
||||
category=ScrapyDeprecationWarning,
|
||||
stacklevel=2
|
||||
)
|
||||
access_key = settings['AWS_ACCESS_KEY_ID']
|
||||
secret_key = settings['AWS_SECRET_ACCESS_KEY']
|
||||
# END Backwards compatibility
|
||||
u = urlparse(uri)
|
||||
self.bucketname = u.hostname
|
||||
self.access_key = u.username or settings['AWS_ACCESS_KEY_ID']
|
||||
self.secret_key = u.password or settings['AWS_SECRET_ACCESS_KEY']
|
||||
self.access_key = u.username or access_key
|
||||
self.secret_key = u.password or secret_key
|
||||
self.is_botocore = is_botocore()
|
||||
self.keyname = u.path[1:] # remove first "/"
|
||||
if self.is_botocore:
|
||||
|
|
@ -111,6 +128,11 @@ class S3FeedStorage(BlockingFeedStorage):
|
|||
import boto
|
||||
self.connect_s3 = boto.connect_s3
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, uri):
|
||||
return cls(uri, crawler.settings['AWS_ACCESS_KEY_ID'],
|
||||
crawler.settings['AWS_SECRET_ACCESS_KEY'])
|
||||
|
||||
def _store_in_thread(self, file):
|
||||
file.seek(0)
|
||||
if self.is_botocore:
|
||||
|
|
@ -181,6 +203,7 @@ class FeedExporter(object):
|
|||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
o = cls(crawler.settings)
|
||||
o.crawler = crawler
|
||||
crawler.signals.connect(o.open_spider, signals.spider_opened)
|
||||
crawler.signals.connect(o.close_spider, signals.spider_closed)
|
||||
crawler.signals.connect(o.item_scraped, signals.item_scraped)
|
||||
|
|
@ -246,18 +269,24 @@ class FeedExporter(object):
|
|||
try:
|
||||
self._get_storage(uri)
|
||||
return True
|
||||
except NotConfigured:
|
||||
logger.error("Disabled feed storage scheme: %(scheme)s",
|
||||
{'scheme': scheme})
|
||||
except NotConfigured as e:
|
||||
logger.error("Disabled feed storage scheme: %(scheme)s. "
|
||||
"Reason: %(reason)s",
|
||||
{'scheme': scheme, 'reason': str(e)})
|
||||
else:
|
||||
logger.error("Unknown feed storage scheme: %(scheme)s",
|
||||
{'scheme': scheme})
|
||||
|
||||
def _get_instance(self, objcls, *args, **kwargs):
|
||||
return create_instance(
|
||||
objcls, self.settings, getattr(self, 'crawler', None),
|
||||
*args, **kwargs)
|
||||
|
||||
def _get_exporter(self, *args, **kwargs):
|
||||
return self.exporters[self.format](*args, **kwargs)
|
||||
return self._get_instance(self.exporters[self.format], *args, **kwargs)
|
||||
|
||||
def _get_storage(self, uri):
|
||||
return self.storages[urlparse(uri).scheme](uri)
|
||||
return self._get_instance(self.storages[urlparse(uri).scheme], uri)
|
||||
|
||||
def _get_uri_params(self, spider):
|
||||
params = {}
|
||||
|
|
|
|||
|
|
@ -91,7 +91,7 @@ class Request(object_ref):
|
|||
"""Create a new Request with the same attributes except for those
|
||||
given new values.
|
||||
"""
|
||||
for x in ['url', 'method', 'headers', 'body', 'cookies', 'meta',
|
||||
for x in ['url', 'method', 'headers', 'body', 'cookies', 'meta', 'flags',
|
||||
'encoding', 'priority', 'dont_filter', 'callback', 'errback']:
|
||||
kwargs.setdefault(x, getattr(self, x))
|
||||
cls = kwargs.pop('cls', self.__class__)
|
||||
|
|
|
|||
|
|
@ -114,10 +114,12 @@ def _get_form(response, formname, formid, formnumber, formxpath):
|
|||
|
||||
def _get_inputs(form, formdata, dont_click, clickdata, response):
|
||||
try:
|
||||
formdata = dict(formdata or ())
|
||||
formdata_keys = dict(formdata or ()).keys()
|
||||
except (ValueError, TypeError):
|
||||
raise ValueError('formdata should be a dict or iterable of tuples')
|
||||
|
||||
if not formdata:
|
||||
formdata = ()
|
||||
inputs = form.xpath('descendant::textarea'
|
||||
'|descendant::select'
|
||||
'|descendant::input[not(@type) or @type['
|
||||
|
|
@ -128,14 +130,17 @@ def _get_inputs(form, formdata, dont_click, clickdata, response):
|
|||
"re": "http://exslt.org/regular-expressions"})
|
||||
values = [(k, u'' if v is None else v)
|
||||
for k, v in (_value(e) for e in inputs)
|
||||
if k and k not in formdata]
|
||||
if k and k not in formdata_keys]
|
||||
|
||||
if not dont_click:
|
||||
clickable = _get_clickable(clickdata, form)
|
||||
if clickable and clickable[0] not in formdata and not clickable[0] is None:
|
||||
values.append(clickable)
|
||||
|
||||
values.extend((k, v) for k, v in formdata.items() if v is not None)
|
||||
if isinstance(formdata, dict):
|
||||
formdata = formdata.items()
|
||||
|
||||
values.extend((k, v) for k, v in formdata if v is not None)
|
||||
return values
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -141,7 +141,7 @@ class SgmlLinkExtractor(FilteringLinkExtractor):
|
|||
base_url = get_base_url(response)
|
||||
body = u''.join(f
|
||||
for x in self.restrict_xpaths
|
||||
for f in response.xpath(x).extract()
|
||||
for f in response.xpath(x).getall()
|
||||
).encode(response.encoding, errors='xmlcharrefreplace')
|
||||
else:
|
||||
body = response.body
|
||||
|
|
|
|||
|
|
@ -181,7 +181,7 @@ class ItemLoader(object):
|
|||
def _get_xpathvalues(self, xpaths, **kw):
|
||||
self._check_selector_method()
|
||||
xpaths = arg_to_iter(xpaths)
|
||||
return flatten(self.selector.xpath(xpath).extract() for xpath in xpaths)
|
||||
return flatten(self.selector.xpath(xpath).getall() for xpath in xpaths)
|
||||
|
||||
def add_css(self, field_name, css, *processors, **kw):
|
||||
values = self._get_cssvalues(css, **kw)
|
||||
|
|
@ -198,6 +198,6 @@ class ItemLoader(object):
|
|||
def _get_cssvalues(self, csss, **kw):
|
||||
self._check_selector_method()
|
||||
csss = arg_to_iter(csss)
|
||||
return flatten(self.selector.css(css).extract() for css in csss)
|
||||
return flatten(self.selector.css(css).getall() for css in csss)
|
||||
|
||||
XPathItemLoader = create_deprecated_class('XPathItemLoader', ItemLoader)
|
||||
|
|
|
|||
|
|
@ -3,7 +3,7 @@ import logging
|
|||
import pprint
|
||||
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.defer import process_parallel, process_chain, process_chain_both
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
@ -32,12 +32,7 @@ class MiddlewareManager(object):
|
|||
for clspath in mwlist:
|
||||
try:
|
||||
mwcls = load_object(clspath)
|
||||
if crawler and hasattr(mwcls, 'from_crawler'):
|
||||
mw = mwcls.from_crawler(crawler)
|
||||
elif hasattr(mwcls, 'from_settings'):
|
||||
mw = mwcls.from_settings(settings)
|
||||
else:
|
||||
mw = mwcls()
|
||||
mw = create_instance(mwcls, settings, crawler)
|
||||
middlewares.append(mw)
|
||||
enabled.append(clspath)
|
||||
except NotConfigured as e:
|
||||
|
|
|
|||
|
|
@ -27,6 +27,10 @@ def _response_from_text(text, st):
|
|||
|
||||
|
||||
class SelectorList(_ParselSelector.selectorlist_cls, object_ref):
|
||||
"""
|
||||
The :class:`SelectorList` class is a subclass of the builtin ``list``
|
||||
class, which provides a few additional methods.
|
||||
"""
|
||||
@deprecated(use_instead='.extract()')
|
||||
def extract_unquoted(self):
|
||||
return [x.extract_unquoted() for x in self]
|
||||
|
|
@ -41,6 +45,35 @@ class SelectorList(_ParselSelector.selectorlist_cls, object_ref):
|
|||
|
||||
|
||||
class Selector(_ParselSelector, object_ref):
|
||||
"""
|
||||
An instance of :class:`Selector` is a wrapper over response to select
|
||||
certain parts of its content.
|
||||
|
||||
``response`` is an :class:`~scrapy.http.HtmlResponse` or an
|
||||
:class:`~scrapy.http.XmlResponse` object that will be used for selecting
|
||||
and extracting data.
|
||||
|
||||
``text`` is a unicode string or utf-8 encoded text for cases when a
|
||||
``response`` isn't available. Using ``text`` and ``response`` together is
|
||||
undefined behavior.
|
||||
|
||||
``type`` defines the selector type, it can be ``"html"``, ``"xml"``
|
||||
or ``None`` (default).
|
||||
|
||||
If ``type`` is ``None``, the selector automatically chooses the best type
|
||||
based on ``response`` type (see below), or defaults to ``"html"`` in case it
|
||||
is used together with ``text``.
|
||||
|
||||
If ``type`` is ``None`` and a ``response`` is passed, the selector type is
|
||||
inferred from the response type as follows:
|
||||
|
||||
* ``"html"`` for :class:`~scrapy.http.HtmlResponse` type
|
||||
* ``"xml"`` for :class:`~scrapy.http.XmlResponse` type
|
||||
* ``"html"`` for anything else
|
||||
|
||||
Otherwise, if ``type`` is set, the selector type will be forced and no
|
||||
detection will occur.
|
||||
"""
|
||||
|
||||
__slots__ = ['response']
|
||||
selectorlist_cls = SelectorList
|
||||
|
|
|
|||
|
|
@ -1,13 +1,10 @@
|
|||
import six
|
||||
import json
|
||||
import copy
|
||||
import warnings
|
||||
from collections import MutableMapping
|
||||
from importlib import import_module
|
||||
from pprint import pformat
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
|
||||
from . import default_settings
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -13,6 +13,7 @@ spider_closed = object()
|
|||
spider_error = object()
|
||||
request_scheduled = object()
|
||||
request_dropped = object()
|
||||
request_reached_downloader = object()
|
||||
response_received = object()
|
||||
response_downloaded = object()
|
||||
item_scraped = object()
|
||||
|
|
|
|||
|
|
@ -3,7 +3,6 @@ Offsite Spider Middleware
|
|||
|
||||
See documentation in docs/topics/spider-middleware.rst
|
||||
"""
|
||||
|
||||
import re
|
||||
import logging
|
||||
import warnings
|
||||
|
|
@ -35,8 +34,9 @@ class OffsiteMiddleware(object):
|
|||
domain = urlparse_cached(x).hostname
|
||||
if domain and domain not in self.domains_seen:
|
||||
self.domains_seen.add(domain)
|
||||
logger.debug("Filtered offsite request to %(domain)r: %(request)s",
|
||||
{'domain': domain, 'request': x}, extra={'spider': spider})
|
||||
logger.debug(
|
||||
"Filtered offsite request to %(domain)r: %(request)s",
|
||||
{'domain': domain, 'request': x}, extra={'spider': spider})
|
||||
self.stats.inc_value('offsite/domains', spider=spider)
|
||||
self.stats.inc_value('offsite/filtered', spider=spider)
|
||||
else:
|
||||
|
|
@ -52,13 +52,15 @@ class OffsiteMiddleware(object):
|
|||
"""Override this method to implement a different offsite policy"""
|
||||
allowed_domains = getattr(spider, 'allowed_domains', None)
|
||||
if not allowed_domains:
|
||||
return re.compile('') # allow all by default
|
||||
return re.compile('') # allow all by default
|
||||
url_pattern = re.compile("^https?://.*$")
|
||||
for domain in allowed_domains:
|
||||
if url_pattern.match(domain):
|
||||
warnings.warn("allowed_domains accepts only domains, not URLs. Ignoring URL entry %s in allowed_domains." % domain, URLWarning)
|
||||
|
||||
regex = r'^(.*\.)?(%s)$' % '|'.join(re.escape(d) for d in allowed_domains if d is not None)
|
||||
message = ("allowed_domains accepts only domains, not URLs. "
|
||||
"Ignoring URL entry %s in allowed_domains." % domain)
|
||||
warnings.warn(message, URLWarning)
|
||||
domains = [re.escape(d) for d in allowed_domains if d is not None]
|
||||
regex = r'^(.*\.)?(%s)$' % '|'.join(domains)
|
||||
return re.compile(regex)
|
||||
|
||||
def spider_opened(self, spider):
|
||||
|
|
|
|||
|
|
@ -14,8 +14,8 @@ class $classname(CrawlSpider):
|
|||
)
|
||||
|
||||
def parse_item(self, response):
|
||||
i = {}
|
||||
#i['domain_id'] = response.xpath('//input[@id="sid"]/@value').extract()
|
||||
#i['name'] = response.xpath('//div[@id="name"]').extract()
|
||||
#i['description'] = response.xpath('//div[@id="description"]').extract()
|
||||
return i
|
||||
item = {}
|
||||
#item['domain_id'] = response.xpath('//input[@id="sid"]/@value').get()
|
||||
#item['name'] = response.xpath('//div[@id="name"]').get()
|
||||
#item['description'] = response.xpath('//div[@id="description"]').get()
|
||||
return item
|
||||
|
|
|
|||
|
|
@ -10,8 +10,8 @@ class $classname(XMLFeedSpider):
|
|||
itertag = 'item' # change it accordingly
|
||||
|
||||
def parse_node(self, response, selector):
|
||||
i = {}
|
||||
#i['url'] = selector.select('url').extract()
|
||||
#i['name'] = selector.select('name').extract()
|
||||
#i['description'] = selector.select('description').extract()
|
||||
return i
|
||||
item = {}
|
||||
#item['url'] = selector.select('url').get()
|
||||
#item['name'] = selector.select('name').get()
|
||||
#item['description'] = selector.select('description').get()
|
||||
return item
|
||||
|
|
|
|||
|
|
@ -117,3 +117,28 @@ def md5sum(file):
|
|||
def rel_has_nofollow(rel):
|
||||
"""Return True if link rel attribute has nofollow type"""
|
||||
return True if rel is not None and 'nofollow' in rel.split() else False
|
||||
|
||||
|
||||
def create_instance(objcls, settings, crawler, *args, **kwargs):
|
||||
"""Construct a class instance using its ``from_crawler`` or
|
||||
``from_settings`` constructors, if available.
|
||||
|
||||
At least one of ``settings`` and ``crawler`` needs to be different from
|
||||
``None``. If ``settings `` is ``None``, ``crawler.settings`` will be used.
|
||||
If ``crawler`` is ``None``, only the ``from_settings`` constructor will be
|
||||
tried.
|
||||
|
||||
``*args`` and ``**kwargs`` are forwarded to the constructors.
|
||||
|
||||
Raises ``ValueError`` if both ``settings`` and ``crawler`` are ``None``.
|
||||
"""
|
||||
if settings is None:
|
||||
if crawler is None:
|
||||
raise ValueError("Specifiy at least one of settings and crawler.")
|
||||
settings = crawler.settings
|
||||
if crawler and hasattr(objcls, 'from_crawler'):
|
||||
return objcls.from_crawler(crawler, *args, **kwargs)
|
||||
elif hasattr(objcls, 'from_settings'):
|
||||
return objcls.from_settings(settings, *args, **kwargs)
|
||||
else:
|
||||
return objcls(*args, **kwargs)
|
||||
|
|
|
|||
2
setup.py
|
|
@ -71,7 +71,7 @@ setup(
|
|||
'pyOpenSSL',
|
||||
'cssselect>=0.9',
|
||||
'six>=1.5.2',
|
||||
'parsel>=1.4',
|
||||
'parsel>=1.5',
|
||||
'PyDispatcher>=2.0.5',
|
||||
'service_identity',
|
||||
],
|
||||
|
|
|
|||
|
|
@ -209,7 +209,7 @@ class MockServer():
|
|||
time.sleep(0.2)
|
||||
|
||||
def url(self, path, is_secure=False):
|
||||
host = self.http_address
|
||||
host = self.http_address.replace('0.0.0.0', '127.0.0.1')
|
||||
if is_secure:
|
||||
host = self.https_address
|
||||
return host + path
|
||||
|
|
|
|||
|
|
@ -3,9 +3,10 @@ pytest-twisted
|
|||
pytest-cov==2.5.1
|
||||
testfixtures
|
||||
jmespath
|
||||
leveldb
|
||||
leveldb; sys_platform != "win32"
|
||||
botocore
|
||||
# optional for shell wrapper tests
|
||||
bpython
|
||||
ipython
|
||||
brotlipy
|
||||
pywin32; sys_platform == "win32"
|
||||
|
|
|
|||
|
|
@ -52,7 +52,8 @@ class CmdlineTest(unittest.TestCase):
|
|||
stats.print_stats()
|
||||
out.seek(0)
|
||||
stats = out.read()
|
||||
self.assertIn('scrapy/commands/version.py', stats)
|
||||
self.assertIn(os.path.join('scrapy', 'commands', 'version.py'),
|
||||
stats)
|
||||
self.assertIn('tottime', stats)
|
||||
finally:
|
||||
shutil.rmtree(path)
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
import os
|
||||
from os.path import join, abspath
|
||||
from twisted.trial import unittest
|
||||
from twisted.internet import defer
|
||||
|
|
@ -7,6 +8,11 @@ from scrapy.utils.python import to_native_str
|
|||
from tests.test_commands import CommandTest
|
||||
|
||||
|
||||
def _textmode(bstr):
|
||||
"""Normalize input the same as writing to a file
|
||||
and reading from it in text mode"""
|
||||
return to_native_str(bstr).replace(os.linesep, '\n')
|
||||
|
||||
class ParseCommandTest(ProcessTest, SiteTest, CommandTest):
|
||||
command = 'parse'
|
||||
|
||||
|
|
@ -97,7 +103,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
'-a', 'test_arg=1',
|
||||
'-c', 'parse',
|
||||
self.url('/html')])
|
||||
self.assertIn("DEBUG: It Works!", to_native_str(stderr))
|
||||
self.assertIn("DEBUG: It Works!", _textmode(stderr))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_request_with_meta(self):
|
||||
|
|
@ -106,13 +112,13 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
'--meta', raw_json_string,
|
||||
'-c', 'parse_request_with_meta',
|
||||
self.url('/html')])
|
||||
self.assertIn("DEBUG: It Works!", to_native_str(stderr))
|
||||
self.assertIn("DEBUG: It Works!", _textmode(stderr))
|
||||
|
||||
_, _, stderr = yield self.execute(['--spider', self.spider_name,
|
||||
'-m', raw_json_string,
|
||||
'-c', 'parse_request_with_meta',
|
||||
self.url('/html')])
|
||||
self.assertIn("DEBUG: It Works!", to_native_str(stderr))
|
||||
self.assertIn("DEBUG: It Works!", _textmode(stderr))
|
||||
|
||||
|
||||
@defer.inlineCallbacks
|
||||
|
|
@ -120,7 +126,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
_, _, stderr = yield self.execute(['--spider', self.spider_name,
|
||||
'-c', 'parse_request_without_meta',
|
||||
self.url('/html')])
|
||||
self.assertIn("DEBUG: It Works!", to_native_str(stderr))
|
||||
self.assertIn("DEBUG: It Works!", _textmode(stderr))
|
||||
|
||||
|
||||
@defer.inlineCallbacks
|
||||
|
|
@ -129,29 +135,29 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
'--pipelines',
|
||||
'-c', 'parse',
|
||||
self.url('/html')])
|
||||
self.assertIn("INFO: It Works!", to_native_str(stderr))
|
||||
self.assertIn("INFO: It Works!", _textmode(stderr))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_parse_items(self):
|
||||
status, out, stderr = yield self.execute(
|
||||
['--spider', self.spider_name, '-c', 'parse', self.url('/html')]
|
||||
)
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", to_native_str(out))
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_parse_items_no_callback_passed(self):
|
||||
status, out, stderr = yield self.execute(
|
||||
['--spider', self.spider_name, self.url('/html')]
|
||||
)
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", to_native_str(out))
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_wrong_callback_passed(self):
|
||||
status, out, stderr = yield self.execute(
|
||||
['--spider', self.spider_name, '-c', 'dummy', self.url('/html')]
|
||||
)
|
||||
self.assertRegexpMatches(to_native_str(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertIn("""Cannot find callback""", to_native_str(stderr))
|
||||
self.assertRegexpMatches(_textmode(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertIn("""Cannot find callback""", _textmode(stderr))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawlspider_matching_rule_callback_set(self):
|
||||
|
|
@ -159,7 +165,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
status, out, stderr = yield self.execute(
|
||||
['--spider', 'goodcrawl'+self.spider_name, '-r', self.url('/html')]
|
||||
)
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", to_native_str(out))
|
||||
self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawlspider_matching_rule_default_callback(self):
|
||||
|
|
@ -167,7 +173,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
status, out, stderr = yield self.execute(
|
||||
['--spider', 'goodcrawl'+self.spider_name, '-r', self.url('/text')]
|
||||
)
|
||||
self.assertIn("""[{}, {'nomatch': 'default'}]""", to_native_str(out))
|
||||
self.assertIn("""[{}, {'nomatch': 'default'}]""", _textmode(out))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_spider_with_no_rules_attribute(self):
|
||||
|
|
@ -175,15 +181,15 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
status, out, stderr = yield self.execute(
|
||||
['--spider', self.spider_name, '-r', self.url('/html')]
|
||||
)
|
||||
self.assertRegexpMatches(to_native_str(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertIn("""No CrawlSpider rules found""", to_native_str(stderr))
|
||||
self.assertRegexpMatches(_textmode(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertIn("""No CrawlSpider rules found""", _textmode(stderr))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawlspider_missing_callback(self):
|
||||
status, out, stderr = yield self.execute(
|
||||
['--spider', 'badcrawl'+self.spider_name, '-r', self.url('/html')]
|
||||
)
|
||||
self.assertRegexpMatches(to_native_str(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertRegexpMatches(_textmode(out), """# Scraped Items -+\n\[\]""")
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawlspider_no_matching_rule(self):
|
||||
|
|
@ -191,5 +197,5 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
|
|||
status, out, stderr = yield self.execute(
|
||||
['--spider', 'badcrawl'+self.spider_name, '-r', self.url('/enc-gb18030')]
|
||||
)
|
||||
self.assertRegexpMatches(to_native_str(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertIn("""Cannot find a rule that matches""", to_native_str(stderr))
|
||||
self.assertRegexpMatches(_textmode(out), """# Scraped Items -+\n\[\]""")
|
||||
self.assertIn("""Cannot find a rule that matches""", _textmode(stderr))
|
||||
|
|
|
|||
|
|
@ -35,7 +35,7 @@ class ShellTest(ProcessTest, SiteTest, unittest.TestCase):
|
|||
|
||||
@defer.inlineCallbacks
|
||||
def test_response_selector_html(self):
|
||||
xpath = 'response.xpath("//p[@class=\'one\']/text()").extract()[0]'
|
||||
xpath = 'response.xpath("//p[@class=\'one\']/text()").get()'
|
||||
_, out, _ = yield self.execute([self.url('/html'), '-c', xpath])
|
||||
self.assertEqual(out.strip(), b'Works')
|
||||
|
||||
|
|
|
|||
|
|
@ -3,18 +3,17 @@ import os
|
|||
import sys
|
||||
import subprocess
|
||||
import tempfile
|
||||
from time import sleep
|
||||
from os.path import exists, join, abspath
|
||||
from shutil import rmtree, copytree
|
||||
from tempfile import mkdtemp
|
||||
from contextlib import contextmanager
|
||||
from threading import Timer
|
||||
|
||||
from twisted.trial import unittest
|
||||
from twisted.internet import defer
|
||||
|
||||
import scrapy
|
||||
from scrapy.utils.python import to_native_str
|
||||
from scrapy.utils.python import retry_on_eintr
|
||||
from scrapy.utils.test import get_testenv
|
||||
from scrapy.utils.testsite import SiteTest
|
||||
from scrapy.utils.testproc import ProcessTest
|
||||
|
|
@ -46,16 +45,18 @@ class ProjectTest(unittest.TestCase):
|
|||
stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||
**popen_kwargs)
|
||||
|
||||
waited = 0
|
||||
interval = 0.2
|
||||
while p.poll() is None:
|
||||
sleep(interval)
|
||||
waited += interval
|
||||
if waited > 15:
|
||||
p.kill()
|
||||
assert False, 'Command took too much time to complete'
|
||||
def kill_proc():
|
||||
p.kill()
|
||||
assert False, 'Command took too much time to complete'
|
||||
|
||||
return p
|
||||
timer = Timer(15, kill_proc)
|
||||
try:
|
||||
timer.start()
|
||||
stdout, stderr = p.communicate()
|
||||
finally:
|
||||
timer.cancel()
|
||||
|
||||
return p, to_native_str(stdout), to_native_str(stderr)
|
||||
|
||||
|
||||
class StartprojectTest(ProjectTest):
|
||||
|
|
@ -111,9 +112,9 @@ class StartprojectTemplatesTest(ProjectTest):
|
|||
assert exists(join(self.tmpl_proj, 'root_template'))
|
||||
|
||||
args = ['--set', 'TEMPLATES_DIR=%s' % self.tmpl]
|
||||
p = self.proc('startproject', self.project_name, *args)
|
||||
out = to_native_str(retry_on_eintr(p.stdout.read))
|
||||
self.assertIn("New Scrapy project %r, using template directory" % self.project_name, out)
|
||||
p, out, err = self.proc('startproject', self.project_name, *args)
|
||||
self.assertIn("New Scrapy project '%s', using template directory"
|
||||
% self.project_name, out)
|
||||
self.assertIn(self.tmpl_proj, out)
|
||||
assert exists(join(self.proj_path, 'root_template'))
|
||||
|
||||
|
|
@ -140,12 +141,10 @@ class GenspiderCommandTest(CommandTest):
|
|||
def test_template(self, tplname='crawl'):
|
||||
args = ['--template=%s' % tplname] if tplname else []
|
||||
spname = 'test_spider'
|
||||
p = self.proc('genspider', spname, 'test.com', *args)
|
||||
out = to_native_str(retry_on_eintr(p.stdout.read))
|
||||
p, out, err = self.proc('genspider', spname, 'test.com', *args)
|
||||
self.assertIn("Created spider %r using template %r in module" % (spname, tplname), out)
|
||||
self.assertTrue(exists(join(self.proj_mod_path, 'spiders', 'test_spider.py')))
|
||||
p = self.proc('genspider', spname, 'test.com', *args)
|
||||
out = to_native_str(retry_on_eintr(p.stdout.read))
|
||||
p, out, err = self.proc('genspider', spname, 'test.com', *args)
|
||||
self.assertIn("Spider %r already exists in module" % spname, out)
|
||||
|
||||
def test_template_basic(self):
|
||||
|
|
@ -212,8 +211,8 @@ class MySpider(scrapy.Spider):
|
|||
return self.proc('runspider', fname, *args)
|
||||
|
||||
def get_log(self, code, name='myspider.py', args=()):
|
||||
p = self.runspider(code, name=name, args=args)
|
||||
return to_native_str(p.stderr.read())
|
||||
p, stdout, stderr = self.runspider(code, name=name, args=args)
|
||||
return stderr
|
||||
|
||||
def test_runspider(self):
|
||||
log = self.get_log(self.debug_log_spider)
|
||||
|
|
@ -223,12 +222,12 @@ class MySpider(scrapy.Spider):
|
|||
self.assertIn("INFO: Spider closed (finished)", log)
|
||||
|
||||
def test_run_fail_spider(self):
|
||||
proc = self.runspider("import scrapy\n" + inspect.getsource(ExceptionSpider))
|
||||
proc, _, _ = self.runspider("import scrapy\n" + inspect.getsource(ExceptionSpider))
|
||||
ret = proc.returncode
|
||||
self.assertNotEqual(ret, 0)
|
||||
|
||||
def test_run_good_spider(self):
|
||||
proc = self.runspider("import scrapy\n" + inspect.getsource(NoRequestsSpider))
|
||||
proc, _, _ = self.runspider("import scrapy\n" + inspect.getsource(NoRequestsSpider))
|
||||
ret = proc.returncode
|
||||
self.assertEqual(ret, 0)
|
||||
|
||||
|
|
@ -279,8 +278,7 @@ class MySpider(scrapy.Spider):
|
|||
self.assertIn("No spider found in file", log)
|
||||
|
||||
def test_runspider_file_not_found(self):
|
||||
p = self.proc('runspider', 'some_non_existent_file')
|
||||
log = to_native_str(p.stderr.read())
|
||||
_, _, log = self.proc('runspider', 'some_non_existent_file')
|
||||
self.assertIn("File not found: some_non_existent_file", log)
|
||||
|
||||
def test_runspider_unable_to_load(self):
|
||||
|
|
@ -304,8 +302,7 @@ class BadSpider(scrapy.Spider):
|
|||
class BenchCommandTest(CommandTest):
|
||||
|
||||
def test_run(self):
|
||||
p = self.proc('bench', '-s', 'LOGSTATS_INTERVAL=0.001',
|
||||
'-s', 'CLOSESPIDER_TIMEOUT=0.01')
|
||||
log = to_native_str(p.stderr.read())
|
||||
_, _, log = self.proc('bench', '-s', 'LOGSTATS_INTERVAL=0.001',
|
||||
'-s', 'CLOSESPIDER_TIMEOUT=0.01')
|
||||
self.assertIn('INFO: Crawled', log)
|
||||
self.assertNotIn('Unhandled Error', log)
|
||||
|
|
|
|||
|
|
@ -1,16 +1,23 @@
|
|||
from unittest import TextTestResult
|
||||
|
||||
from six import get_unbound_function
|
||||
from twisted.internet import defer
|
||||
from twisted.python import failure
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy import FormRequest
|
||||
from scrapy.crawler import CrawlerRunner
|
||||
from scrapy.spidermiddlewares.httperror import HttpError
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.http import Request
|
||||
from scrapy.item import Item, Field
|
||||
from scrapy.contracts import ContractsManager
|
||||
from scrapy.contracts import ContractsManager, Contract
|
||||
from scrapy.contracts.default import (
|
||||
UrlContract,
|
||||
ReturnsContract,
|
||||
ScrapesContract,
|
||||
)
|
||||
from tests.mockserver import MockServer
|
||||
|
||||
|
||||
class TestItem(Item):
|
||||
|
|
@ -22,6 +29,30 @@ class ResponseMock(object):
|
|||
url = 'http://scrapy.org'
|
||||
|
||||
|
||||
class CustomSuccessContract(Contract):
|
||||
name = 'custom_success_contract'
|
||||
|
||||
def adjust_request_args(self, args):
|
||||
args['url'] = 'http://scrapy.org'
|
||||
return args
|
||||
|
||||
|
||||
class CustomFailContract(Contract):
|
||||
name = 'custom_fail_contract'
|
||||
|
||||
def adjust_request_args(self, args):
|
||||
raise TypeError('Error in adjust_request_args')
|
||||
|
||||
|
||||
class CustomFormContract(Contract):
|
||||
name = 'custom_form'
|
||||
request_cls = FormRequest
|
||||
|
||||
def adjust_request_args(self, args):
|
||||
args['formdata'] = {'name': 'scrapy'}
|
||||
return args
|
||||
|
||||
|
||||
class TestSpider(Spider):
|
||||
name = 'demo_spider'
|
||||
|
||||
|
|
@ -98,9 +129,47 @@ class TestSpider(Spider):
|
|||
"""
|
||||
pass
|
||||
|
||||
def custom_form(self, response):
|
||||
"""
|
||||
@url http://scrapy.org
|
||||
@custom_form
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class CustomContractSuccessSpider(Spider):
|
||||
name = 'custom_contract_success_spider'
|
||||
|
||||
def parse(self, response):
|
||||
"""
|
||||
@custom_success_contract
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class CustomContractFailSpider(Spider):
|
||||
name = 'custom_contract_fail_spider'
|
||||
|
||||
def parse(self, response):
|
||||
"""
|
||||
@custom_fail_contract
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class InheritsTestSpider(TestSpider):
|
||||
name = 'inherits_demo_spider'
|
||||
|
||||
|
||||
class ContractsManagerTest(unittest.TestCase):
|
||||
contracts = [UrlContract, ReturnsContract, ScrapesContract]
|
||||
contracts = [
|
||||
UrlContract,
|
||||
ReturnsContract,
|
||||
ScrapesContract,
|
||||
CustomFormContract,
|
||||
CustomSuccessContract,
|
||||
CustomFailContract,
|
||||
]
|
||||
|
||||
def setUp(self):
|
||||
self.conman = ContractsManager(self.contracts)
|
||||
|
|
@ -114,6 +183,9 @@ class ContractsManagerTest(unittest.TestCase):
|
|||
self.assertTrue(self.results.failures)
|
||||
self.assertFalse(self.results.errors)
|
||||
|
||||
def should_error(self):
|
||||
self.assertTrue(self.results.errors)
|
||||
|
||||
def test_contracts(self):
|
||||
spider = TestSpider()
|
||||
|
||||
|
|
@ -175,13 +247,77 @@ class ContractsManagerTest(unittest.TestCase):
|
|||
self.should_succeed()
|
||||
|
||||
# scrapes_item_fail
|
||||
request = self.conman.from_method(spider.scrapes_item_fail,
|
||||
self.results)
|
||||
request = self.conman.from_method(spider.scrapes_item_fail, self.results)
|
||||
request.callback(response)
|
||||
self.should_fail()
|
||||
|
||||
# scrapes_dict_item_fail
|
||||
request = self.conman.from_method(spider.scrapes_dict_item_fail,
|
||||
self.results)
|
||||
request = self.conman.from_method(spider.scrapes_dict_item_fail, self.results)
|
||||
request.callback(response)
|
||||
self.should_fail()
|
||||
|
||||
def test_custom_contracts(self):
|
||||
self.conman.from_spider(CustomContractSuccessSpider(), self.results)
|
||||
self.should_succeed()
|
||||
|
||||
self.conman.from_spider(CustomContractFailSpider(), self.results)
|
||||
self.should_error()
|
||||
|
||||
def test_errback(self):
|
||||
spider = TestSpider()
|
||||
response = ResponseMock()
|
||||
|
||||
try:
|
||||
raise HttpError(response, 'Ignoring non-200 response')
|
||||
except HttpError:
|
||||
failure_mock = failure.Failure()
|
||||
|
||||
request = self.conman.from_method(spider.returns_request, self.results)
|
||||
request.errback(failure_mock)
|
||||
|
||||
self.assertFalse(self.results.failures)
|
||||
self.assertTrue(self.results.errors)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_same_url(self):
|
||||
|
||||
class TestSameUrlSpider(Spider):
|
||||
name = 'test_same_url'
|
||||
|
||||
def __init__(self, *args, **kwargs):
|
||||
super(TestSameUrlSpider, self).__init__(*args, **kwargs)
|
||||
self.visited = 0
|
||||
|
||||
def start_requests(s):
|
||||
return self.conman.from_spider(s, self.results)
|
||||
|
||||
def parse_first(self, response):
|
||||
self.visited += 1
|
||||
return TestItem()
|
||||
|
||||
def parse_second(self, response):
|
||||
self.visited += 1
|
||||
return TestItem()
|
||||
|
||||
with MockServer() as mockserver:
|
||||
contract_doc = '@url {}'.format(mockserver.url('/status?n=200'))
|
||||
|
||||
get_unbound_function(TestSameUrlSpider.parse_first).__doc__ = contract_doc
|
||||
get_unbound_function(TestSameUrlSpider.parse_second).__doc__ = contract_doc
|
||||
|
||||
crawler = CrawlerRunner().create_crawler(TestSameUrlSpider)
|
||||
yield crawler.crawl()
|
||||
|
||||
self.assertEqual(crawler.spider.visited, 2)
|
||||
|
||||
def test_form_contract(self):
|
||||
spider = TestSpider()
|
||||
request = self.conman.from_method(spider.custom_form, self.results)
|
||||
self.assertEqual(request.method, 'POST')
|
||||
self.assertIsInstance(request, FormRequest)
|
||||
|
||||
def test_inherited_contracts(self):
|
||||
spider = InheritsTestSpider()
|
||||
|
||||
requests = self.conman.from_spider(spider, self.results)
|
||||
self.assertTrue(requests)
|
||||
|
|
|
|||
|
|
@ -1,10 +1,9 @@
|
|||
import logging
|
||||
import tempfile
|
||||
import warnings
|
||||
import unittest
|
||||
|
||||
from twisted.internet import defer
|
||||
import twisted.trial.unittest
|
||||
from twisted.trial import unittest
|
||||
|
||||
import scrapy
|
||||
from scrapy.crawler import Crawler, CrawlerRunner, CrawlerProcess
|
||||
|
|
@ -94,26 +93,29 @@ class CrawlerLoggingTestCase(unittest.TestCase):
|
|||
assert get_scrapy_root_handler() is None
|
||||
|
||||
def test_spider_custom_settings_log_level(self):
|
||||
with tempfile.NamedTemporaryFile() as log_file:
|
||||
class MySpider(scrapy.Spider):
|
||||
name = 'spider'
|
||||
custom_settings = {
|
||||
'LOG_LEVEL': 'INFO',
|
||||
'LOG_FILE': log_file.name,
|
||||
# disable telnet if not available to avoid an extra warning
|
||||
'TELNETCONSOLE_ENABLED': telnet.TWISTED_CONCH_AVAILABLE,
|
||||
}
|
||||
log_file = self.mktemp()
|
||||
class MySpider(scrapy.Spider):
|
||||
name = 'spider'
|
||||
custom_settings = {
|
||||
'LOG_LEVEL': 'INFO',
|
||||
'LOG_FILE': log_file,
|
||||
# disable telnet if not available to avoid an extra warning
|
||||
'TELNETCONSOLE_ENABLED': telnet.TWISTED_CONCH_AVAILABLE,
|
||||
}
|
||||
|
||||
configure_logging()
|
||||
self.assertEqual(get_scrapy_root_handler().level, logging.DEBUG)
|
||||
crawler = Crawler(MySpider, {})
|
||||
self.assertEqual(get_scrapy_root_handler().level, logging.INFO)
|
||||
info_count = crawler.stats.get_value('log_count/INFO')
|
||||
logging.debug('debug message')
|
||||
logging.info('info message')
|
||||
logging.warning('warning message')
|
||||
logging.error('error message')
|
||||
|
||||
with open(log_file, 'rb') as fo:
|
||||
logged = fo.read().decode('utf8')
|
||||
|
||||
configure_logging()
|
||||
self.assertEqual(get_scrapy_root_handler().level, logging.DEBUG)
|
||||
crawler = Crawler(MySpider, {})
|
||||
self.assertEqual(get_scrapy_root_handler().level, logging.INFO)
|
||||
info_count = crawler.stats.get_value('log_count/INFO')
|
||||
logging.debug('debug message')
|
||||
logging.info('info message')
|
||||
logging.warning('warning message')
|
||||
logging.error('error message')
|
||||
logged = log_file.read().decode('utf8')
|
||||
self.assertNotIn('debug message', logged)
|
||||
self.assertIn('info message', logged)
|
||||
self.assertIn('warning message', logged)
|
||||
|
|
@ -141,9 +143,8 @@ class CrawlerRunnerTestCase(BaseCrawlerTest):
|
|||
settings = Settings({
|
||||
'SPIDER_LOADER_CLASS': 'tests.test_crawler.SpiderLoaderWithWrongInterface'
|
||||
})
|
||||
with warnings.catch_warnings(record=True) as w, \
|
||||
self.assertRaises(AttributeError):
|
||||
CrawlerRunner(settings)
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
self.assertRaises(AttributeError, CrawlerRunner, settings)
|
||||
self.assertEqual(len(w), 1)
|
||||
self.assertIn("SPIDER_LOADER_CLASS", str(w[0].message))
|
||||
self.assertIn("scrapy.interfaces.ISpiderLoader", str(w[0].message))
|
||||
|
|
@ -203,7 +204,7 @@ class NoRequestsSpider(scrapy.Spider):
|
|||
return []
|
||||
|
||||
|
||||
class CrawlerRunnerHasSpider(twisted.trial.unittest.TestCase):
|
||||
class CrawlerRunnerHasSpider(unittest.TestCase):
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler_runner_bootstrap_successful(self):
|
||||
|
|
|
|||
|
|
@ -1,7 +1,8 @@
|
|||
import os
|
||||
import six
|
||||
import contextlib
|
||||
import shutil
|
||||
import tempfile
|
||||
import contextlib
|
||||
try:
|
||||
from unittest import mock
|
||||
except ImportError:
|
||||
|
|
@ -913,7 +914,9 @@ class BaseFTPTestCase(unittest.TestCase):
|
|||
return self._add_test_callbacks(d, _test)
|
||||
|
||||
def test_ftp_local_filename(self):
|
||||
local_fname = b"/tmp/file.txt"
|
||||
f, local_fname = tempfile.mkstemp()
|
||||
local_fname = to_bytes(local_fname)
|
||||
os.close(f)
|
||||
meta = {"ftp_local_filename": local_fname}
|
||||
meta.update(self.req_meta)
|
||||
request = Request(url="ftp://127.0.0.1:%s/file.txt" % self.portNum,
|
||||
|
|
@ -922,7 +925,8 @@ class BaseFTPTestCase(unittest.TestCase):
|
|||
|
||||
def _test(r):
|
||||
self.assertEqual(r.body, local_fname)
|
||||
self.assertEqual(r.headers, {b'Local Filename': [b'/tmp/file.txt'], b'Size': [b'17']})
|
||||
self.assertEqual(r.headers, {b'Local Filename': [local_fname],
|
||||
b'Size': [b'17']})
|
||||
self.assertTrue(os.path.exists(local_fname))
|
||||
with open(local_fname, "rb") as f:
|
||||
self.assertEqual(f.read(), b"I have the power!")
|
||||
|
|
|
|||
|
|
@ -5,11 +5,60 @@ import shutil
|
|||
|
||||
from scrapy.dupefilters import RFPDupeFilter
|
||||
from scrapy.http import Request
|
||||
from scrapy.core.scheduler import Scheduler
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.utils.job import job_dir
|
||||
from scrapy.utils.test import get_crawler
|
||||
|
||||
|
||||
class FromCrawlerRFPDupeFilter(RFPDupeFilter):
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
debug = crawler.settings.getbool('DUPEFILTER_DEBUG')
|
||||
df = cls(job_dir(crawler.settings), debug)
|
||||
df.method = 'from_crawler'
|
||||
return df
|
||||
|
||||
|
||||
class FromSettingsRFPDupeFilter(RFPDupeFilter):
|
||||
|
||||
@classmethod
|
||||
def from_settings(cls, settings):
|
||||
debug = settings.getbool('DUPEFILTER_DEBUG')
|
||||
df = cls(job_dir(settings), debug)
|
||||
df.method = 'from_settings'
|
||||
return df
|
||||
|
||||
|
||||
class DirectDupeFilter(object):
|
||||
method = 'n/a'
|
||||
|
||||
|
||||
class RFPDupeFilterTest(unittest.TestCase):
|
||||
|
||||
def test_df_from_crawler_scheduler(self):
|
||||
settings = {'DUPEFILTER_DEBUG': True,
|
||||
'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
scheduler = Scheduler.from_crawler(crawler)
|
||||
self.assertTrue(scheduler.df.debug)
|
||||
self.assertEqual(scheduler.df.method, 'from_crawler')
|
||||
|
||||
def test_df_from_settings_scheduler(self):
|
||||
settings = {'DUPEFILTER_DEBUG': True,
|
||||
'DUPEFILTER_CLASS': __name__ + '.FromSettingsRFPDupeFilter'}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
scheduler = Scheduler.from_crawler(crawler)
|
||||
self.assertTrue(scheduler.df.debug)
|
||||
self.assertEqual(scheduler.df.method, 'from_settings')
|
||||
|
||||
def test_df_direct_scheduler(self):
|
||||
settings = {'DUPEFILTER_CLASS': __name__ + '.DirectDupeFilter'}
|
||||
crawler = get_crawler(settings_dict=settings)
|
||||
scheduler = Scheduler.from_crawler(crawler)
|
||||
self.assertEqual(scheduler.df.method, 'n/a')
|
||||
|
||||
def test_filter(self):
|
||||
dupefilter = RFPDupeFilter()
|
||||
dupefilter.open()
|
||||
|
|
|
|||
|
|
@ -103,6 +103,7 @@ class CrawlerRun(object):
|
|||
self.respplug = []
|
||||
self.reqplug = []
|
||||
self.reqdropped = []
|
||||
self.reqreached = []
|
||||
self.itemerror = []
|
||||
self.itemresp = []
|
||||
self.signals_catched = {}
|
||||
|
|
@ -124,6 +125,7 @@ class CrawlerRun(object):
|
|||
self.crawler.signals.connect(self.item_error, signals.item_error)
|
||||
self.crawler.signals.connect(self.request_scheduled, signals.request_scheduled)
|
||||
self.crawler.signals.connect(self.request_dropped, signals.request_dropped)
|
||||
self.crawler.signals.connect(self.request_reached, signals.request_reached_downloader)
|
||||
self.crawler.signals.connect(self.response_downloaded, signals.response_downloaded)
|
||||
self.crawler.crawl(start_urls=start_urls)
|
||||
self.spider = self.crawler.spider
|
||||
|
|
@ -155,6 +157,9 @@ class CrawlerRun(object):
|
|||
def request_scheduled(self, request, spider):
|
||||
self.reqplug.append((request, spider))
|
||||
|
||||
def request_reached(self, request, spider):
|
||||
self.reqreached.append((request, spider))
|
||||
|
||||
def request_dropped(self, request, spider):
|
||||
self.reqdropped.append((request, spider))
|
||||
|
||||
|
|
@ -212,6 +217,8 @@ class EngineTest(unittest.TestCase):
|
|||
responses_count = len(self.run.respplug)
|
||||
self.assertEqual(scheduled_requests_count,
|
||||
dropped_requests_count + responses_count)
|
||||
self.assertEqual(len(self.run.reqreached),
|
||||
responses_count)
|
||||
|
||||
def _assert_dropped_requests(self):
|
||||
self.assertEqual(len(self.run.reqdropped), 1)
|
||||
|
|
@ -219,6 +226,7 @@ class EngineTest(unittest.TestCase):
|
|||
def _assert_downloaded_responses(self):
|
||||
# response tests
|
||||
self.assertEqual(8, len(self.run.respplug))
|
||||
self.assertEqual(8, len(self.run.reqreached))
|
||||
|
||||
for response, _ in self.run.respplug:
|
||||
if self.run.getpath(response.url) == '/item999.html':
|
||||
|
|
|
|||
|
|
@ -2,20 +2,24 @@ from __future__ import absolute_import
|
|||
import os
|
||||
import csv
|
||||
import json
|
||||
import warnings
|
||||
from io import BytesIO
|
||||
import tempfile
|
||||
import shutil
|
||||
from six.moves.urllib.parse import urlparse
|
||||
from six.moves.urllib.parse import urljoin, urlparse
|
||||
from six.moves.urllib.request import pathname2url
|
||||
|
||||
from zope.interface.verify import verifyObject
|
||||
from twisted.trial import unittest
|
||||
from twisted.internet import defer
|
||||
from scrapy.crawler import CrawlerRunner
|
||||
from scrapy.settings import Settings
|
||||
from tests import mock
|
||||
from tests.mockserver import MockServer
|
||||
from w3lib.url import path_to_file_uri
|
||||
|
||||
import scrapy
|
||||
from scrapy.exporters import CsvItemExporter
|
||||
from scrapy.extensions.feedexport import (
|
||||
IFeedStorage, FileFeedStorage, FTPFeedStorage,
|
||||
S3FeedStorage, StdoutFeedStorage,
|
||||
|
|
@ -130,13 +134,49 @@ class BlockingFeedStorageTest(unittest.TestCase):
|
|||
|
||||
class S3FeedStorageTest(unittest.TestCase):
|
||||
|
||||
@mock.patch('scrapy.conf.settings', new={'AWS_ACCESS_KEY_ID': 'conf_key',
|
||||
'AWS_SECRET_ACCESS_KEY': 'conf_secret'}, create=True)
|
||||
def test_parse_credentials(self):
|
||||
try:
|
||||
import boto
|
||||
except ImportError:
|
||||
raise unittest.SkipTest("S3FeedStorage requires boto")
|
||||
aws_credentials = {'AWS_ACCESS_KEY_ID': 'settings_key',
|
||||
'AWS_SECRET_ACCESS_KEY': 'settings_secret'}
|
||||
crawler = get_crawler(settings_dict=aws_credentials)
|
||||
# Instantiate with crawler
|
||||
storage = S3FeedStorage.from_crawler(crawler,
|
||||
's3://mybucket/export.csv')
|
||||
self.assertEqual(storage.access_key, 'settings_key')
|
||||
self.assertEqual(storage.secret_key, 'settings_secret')
|
||||
# Instantiate directly
|
||||
storage = S3FeedStorage('s3://mybucket/export.csv',
|
||||
aws_credentials['AWS_ACCESS_KEY_ID'],
|
||||
aws_credentials['AWS_SECRET_ACCESS_KEY'])
|
||||
self.assertEqual(storage.access_key, 'settings_key')
|
||||
self.assertEqual(storage.secret_key, 'settings_secret')
|
||||
# URI priority > settings priority
|
||||
storage = S3FeedStorage('s3://uri_key:uri_secret@mybucket/export.csv',
|
||||
aws_credentials['AWS_ACCESS_KEY_ID'],
|
||||
aws_credentials['AWS_SECRET_ACCESS_KEY'])
|
||||
self.assertEqual(storage.access_key, 'uri_key')
|
||||
self.assertEqual(storage.secret_key, 'uri_secret')
|
||||
# Backwards compatibility for initialising without settings
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
storage = S3FeedStorage('s3://mybucket/export.csv')
|
||||
self.assertEqual(storage.access_key, 'conf_key')
|
||||
self.assertEqual(storage.secret_key, 'conf_secret')
|
||||
self.assertTrue('without AWS keys' in str(w[-1].message))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_store(self):
|
||||
assert_aws_environ()
|
||||
uri = os.environ.get('S3_TEST_FILE_URI')
|
||||
if not uri:
|
||||
raise unittest.SkipTest("No S3 URI available for testing")
|
||||
storage = S3FeedStorage(uri)
|
||||
access_key = os.environ.get('AWS_ACCESS_KEY_ID')
|
||||
secret_key = os.environ.get('AWS_SECRET_ACCESS_KEY')
|
||||
storage = S3FeedStorage(uri, access_key, secret_key)
|
||||
verifyObject(IFeedStorage, storage)
|
||||
file = storage.open(scrapy.Spider("default"))
|
||||
expected_content = b"content: \xe2\x98\x83"
|
||||
|
|
@ -159,6 +199,23 @@ class StdoutFeedStorageTest(unittest.TestCase):
|
|||
self.assertEqual(out.getvalue(), b"content")
|
||||
|
||||
|
||||
class FromCrawlerMixin(object):
|
||||
init_with_crawler = False
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, *args, **kwargs):
|
||||
cls.init_with_crawler = True
|
||||
return cls(*args, **kwargs)
|
||||
|
||||
|
||||
class FromCrawlerCsvItemExporter(CsvItemExporter, FromCrawlerMixin):
|
||||
pass
|
||||
|
||||
|
||||
class FromCrawlerFileFeedStorage(FileFeedStorage, FromCrawlerMixin):
|
||||
pass
|
||||
|
||||
|
||||
class FeedExportTest(unittest.TestCase):
|
||||
|
||||
class MyItem(scrapy.Item):
|
||||
|
|
@ -170,9 +227,10 @@ class FeedExportTest(unittest.TestCase):
|
|||
def run_and_export(self, spider_cls, settings=None):
|
||||
""" Run spider with specified settings; return exported data. """
|
||||
tmpdir = tempfile.mkdtemp()
|
||||
res_name = tmpdir + '/res'
|
||||
res_path = os.path.join(tmpdir, 'res')
|
||||
res_uri = urljoin('file:', pathname2url(res_path))
|
||||
defaults = {
|
||||
'FEED_URI': 'file://' + res_name,
|
||||
'FEED_URI': res_uri,
|
||||
'FEED_FORMAT': 'csv',
|
||||
}
|
||||
defaults.update(settings or {})
|
||||
|
|
@ -182,11 +240,13 @@ class FeedExportTest(unittest.TestCase):
|
|||
spider_cls.start_urls = [s.url('/')]
|
||||
yield runner.crawl(spider_cls)
|
||||
|
||||
with open(res_name, 'rb') as f:
|
||||
defer.returnValue(f.read())
|
||||
with open(res_path, 'rb') as f:
|
||||
content = f.read()
|
||||
|
||||
finally:
|
||||
shutil.rmtree(tmpdir)
|
||||
shutil.rmtree(tmpdir, ignore_errors=True)
|
||||
|
||||
defer.returnValue(content)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def exported_data(self, items, settings):
|
||||
|
|
@ -598,3 +658,15 @@ class FeedExportTest(unittest.TestCase):
|
|||
data = yield self.exported_data(items, settings)
|
||||
print(row['format'], row['indent'])
|
||||
self.assertEqual(row['expected'], data)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_init_exporters_storages_with_crawler(self):
|
||||
settings = {
|
||||
'FEED_EXPORTERS': {'csv': 'tests.test_feedexport.'
|
||||
'FromCrawlerCsvItemExporter'},
|
||||
'FEED_STORAGES': {'file': 'tests.test_feedexport.'
|
||||
'FromCrawlerFileFeedStorage'},
|
||||
}
|
||||
yield self.exported_data({}, settings)
|
||||
self.assertTrue(FromCrawlerCsvItemExporter.init_with_crawler)
|
||||
self.assertTrue(FromCrawlerFileFeedStorage.init_with_crawler)
|
||||
|
|
|
|||
|
|
@ -174,7 +174,8 @@ class RequestTest(unittest.TestCase):
|
|||
def somecallback():
|
||||
pass
|
||||
|
||||
r1 = self.request_class("http://www.example.com", callback=somecallback, errback=somecallback)
|
||||
r1 = self.request_class("http://www.example.com", flags=['f1', 'f2'],
|
||||
callback=somecallback, errback=somecallback)
|
||||
r1.meta['foo'] = 'bar'
|
||||
r2 = r1.copy()
|
||||
|
||||
|
|
@ -184,6 +185,10 @@ class RequestTest(unittest.TestCase):
|
|||
assert r2.callback is r1.callback
|
||||
assert r2.errback is r2.errback
|
||||
|
||||
# make sure flags list is shallow copied
|
||||
assert r1.flags is not r2.flags, "flags must be a shallow copy, not identical"
|
||||
self.assertEqual(r1.flags, r2.flags)
|
||||
|
||||
# make sure meta dict is shallow copied
|
||||
assert r1.meta is not r2.meta, "meta must be a shallow copy, not identical"
|
||||
self.assertEqual(r1.meta, r2.meta)
|
||||
|
|
@ -401,6 +406,29 @@ class FormRequestTest(RequestTest):
|
|||
self.assertEqual(fs[u'test2'], [u'xxx µ'])
|
||||
self.assertEqual(fs[u'six'], [u'seven'])
|
||||
|
||||
def test_from_response_duplicate_form_key(self):
|
||||
response = _buildresponse(
|
||||
'<form></form>',
|
||||
url='http://www.example.com')
|
||||
req = self.request_class.from_response(response,
|
||||
method='GET',
|
||||
formdata=(('foo', 'bar'), ('foo', 'baz')))
|
||||
self.assertEqual(urlparse(req.url).hostname, 'www.example.com')
|
||||
self.assertEqual(urlparse(req.url).query, 'foo=bar&foo=baz')
|
||||
|
||||
def test_from_response_override_duplicate_form_key(self):
|
||||
response = _buildresponse(
|
||||
"""<form action="get.php" method="POST">
|
||||
<input type="hidden" name="one" value="1">
|
||||
<input type="hidden" name="two" value="3">
|
||||
</form>""")
|
||||
req = self.request_class.from_response(
|
||||
response,
|
||||
formdata=(('two', '2'), ('two', '4')))
|
||||
fs = _qs(req)
|
||||
self.assertEqual(fs[b'one'], [b'1'])
|
||||
self.assertEqual(fs[b'two'], [b'2', b'4'])
|
||||
|
||||
def test_from_response_extra_headers(self):
|
||||
response = _buildresponse(
|
||||
"""<form action="post.php" method="POST">
|
||||
|
|
|
|||
|
|
@ -336,11 +336,11 @@ class TextResponseTest(BaseResponseTest):
|
|||
self.assertIs(response.selector.response, response)
|
||||
|
||||
self.assertEqual(
|
||||
response.selector.xpath("//title/text()").extract(),
|
||||
response.selector.xpath("//title/text()").getall(),
|
||||
[u'Some page']
|
||||
)
|
||||
self.assertEqual(
|
||||
response.selector.css("title::text").extract(),
|
||||
response.selector.css("title::text").getall(),
|
||||
[u'Some page']
|
||||
)
|
||||
self.assertEqual(
|
||||
|
|
@ -353,12 +353,12 @@ class TextResponseTest(BaseResponseTest):
|
|||
response = self.response_class("http://www.example.com", body=body)
|
||||
|
||||
self.assertEqual(
|
||||
response.xpath("//title/text()").extract(),
|
||||
response.selector.xpath("//title/text()").extract(),
|
||||
response.xpath("//title/text()").getall(),
|
||||
response.selector.xpath("//title/text()").getall(),
|
||||
)
|
||||
self.assertEqual(
|
||||
response.css("title::text").extract(),
|
||||
response.selector.css("title::text").extract(),
|
||||
response.css("title::text").getall(),
|
||||
response.selector.css("title::text").getall(),
|
||||
)
|
||||
|
||||
def test_selector_shortcuts_kwargs(self):
|
||||
|
|
@ -366,13 +366,13 @@ class TextResponseTest(BaseResponseTest):
|
|||
response = self.response_class("http://www.example.com", body=body)
|
||||
|
||||
self.assertEqual(
|
||||
response.xpath("normalize-space(//p[@class=$pclass])", pclass="content").extract(),
|
||||
response.xpath("normalize-space(//p[@class=\"content\"])").extract(),
|
||||
response.xpath("normalize-space(//p[@class=$pclass])", pclass="content").getall(),
|
||||
response.xpath("normalize-space(//p[@class=\"content\"])").getall(),
|
||||
)
|
||||
self.assertEqual(
|
||||
response.xpath("//title[count(following::p[@class=$pclass])=$pcount]/text()",
|
||||
pclass="content", pcount=1).extract(),
|
||||
response.xpath("//title[count(following::p[@class=\"content\"])=1]/text()").extract(),
|
||||
pclass="content", pcount=1).getall(),
|
||||
response.xpath("//title[count(following::p[@class=\"content\"])=1]/text()").getall(),
|
||||
)
|
||||
|
||||
def test_urljoin_with_base_url(self):
|
||||
|
|
@ -562,7 +562,7 @@ class XmlResponseTest(TextResponseTest):
|
|||
self.assertIs(response.selector.response, response)
|
||||
|
||||
self.assertEqual(
|
||||
response.selector.xpath("//elem/text()").extract(),
|
||||
response.selector.xpath("//elem/text()").getall(),
|
||||
[u'value']
|
||||
)
|
||||
|
||||
|
|
@ -571,8 +571,8 @@ class XmlResponseTest(TextResponseTest):
|
|||
response = self.response_class("http://www.example.com", body=body)
|
||||
|
||||
self.assertEqual(
|
||||
response.xpath("//elem/text()").extract(),
|
||||
response.selector.xpath("//elem/text()").extract(),
|
||||
response.xpath("//elem/text()").getall(),
|
||||
response.selector.xpath("//elem/text()").getall(),
|
||||
)
|
||||
|
||||
def test_selector_shortcuts_kwargs(self):
|
||||
|
|
@ -583,12 +583,12 @@ class XmlResponseTest(TextResponseTest):
|
|||
response = self.response_class("http://www.example.com", body=body)
|
||||
|
||||
self.assertEqual(
|
||||
response.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).extract(),
|
||||
response.selector.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).extract(),
|
||||
response.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).getall(),
|
||||
response.selector.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).getall(),
|
||||
)
|
||||
|
||||
response.selector.register_namespace('s2', 'http://scrapy.org')
|
||||
self.assertEqual(
|
||||
response.xpath("//s1:elem/text()", namespaces={'s1': 'http://scrapy.org'}).extract(),
|
||||
response.selector.xpath("//s2:elem/text()").extract(),
|
||||
response.xpath("//s1:elem/text()", namespaces={'s1': 'http://scrapy.org'}).getall(),
|
||||
response.selector.xpath("//s2:elem/text()").getall(),
|
||||
)
|
||||
|
|
|
|||
|
|
@ -634,7 +634,7 @@ class SubselectorLoaderTest(unittest.TestCase):
|
|||
nl = l.nested_xpath("//header")
|
||||
nl.add_xpath('name', 'div/text()')
|
||||
nl.add_css('name_div', '#id')
|
||||
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').extract())
|
||||
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').getall())
|
||||
|
||||
self.assertEqual(l.get_output_value('name'), [u'marta'])
|
||||
self.assertEqual(l.get_output_value('name_div'), [u'<div id="id">marta</div>'])
|
||||
|
|
@ -649,7 +649,7 @@ class SubselectorLoaderTest(unittest.TestCase):
|
|||
nl = l.nested_css("header")
|
||||
nl.add_xpath('name', 'div/text()')
|
||||
nl.add_css('name_div', '#id')
|
||||
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').extract())
|
||||
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').getall())
|
||||
|
||||
self.assertEqual(l.get_output_value('name'), [u'marta'])
|
||||
self.assertEqual(l.get_output_value('name_div'), [u'<div id="id">marta</div>'])
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ class MediaDownloadSpider(SimpleSpider):
|
|||
for href in response.xpath('''
|
||||
//table[thead/tr/th="Filename"]
|
||||
/tbody//a/@href
|
||||
''').extract()],
|
||||
''').getall()],
|
||||
}
|
||||
yield item
|
||||
|
||||
|
|
|
|||
|
|
@ -20,17 +20,17 @@ class SelectorTestCase(unittest.TestCase):
|
|||
for x in xl:
|
||||
assert isinstance(x, Selector)
|
||||
|
||||
self.assertEqual(sel.xpath('//input').extract(),
|
||||
[x.extract() for x in sel.xpath('//input')])
|
||||
self.assertEqual(sel.xpath('//input').getall(),
|
||||
[x.get() for x in sel.xpath('//input')])
|
||||
|
||||
self.assertEqual([x.extract() for x in sel.xpath("//input[@name='a']/@name")],
|
||||
self.assertEqual([x.get() for x in sel.xpath("//input[@name='a']/@name")],
|
||||
[u'a'])
|
||||
self.assertEqual([x.extract() for x in sel.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
|
||||
self.assertEqual([x.get() for x in sel.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
|
||||
[u'12.0'])
|
||||
|
||||
self.assertEqual(sel.xpath("concat('xpath', 'rules')").extract(),
|
||||
self.assertEqual(sel.xpath("concat('xpath', 'rules')").getall(),
|
||||
[u'xpathrules'])
|
||||
self.assertEqual([x.extract() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
|
||||
self.assertEqual([x.get() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
|
||||
[u'12'])
|
||||
|
||||
def test_root_base_url(self):
|
||||
|
|
@ -60,12 +60,12 @@ class SelectorTestCase(unittest.TestCase):
|
|||
text = b'<div><img src="a.jpg"><p>Hello</div>'
|
||||
sel = Selector(XmlResponse('http://example.com', body=text, encoding='utf-8'))
|
||||
self.assertEqual(sel.type, 'xml')
|
||||
self.assertEqual(sel.xpath("//div").extract(),
|
||||
self.assertEqual(sel.xpath("//div").getall(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></img></div>'])
|
||||
|
||||
sel = Selector(HtmlResponse('http://example.com', body=text, encoding='utf-8'))
|
||||
self.assertEqual(sel.type, 'html')
|
||||
self.assertEqual(sel.xpath("//div").extract(),
|
||||
self.assertEqual(sel.xpath("//div").getall(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></div>'])
|
||||
|
||||
def test_http_header_encoding_precedence(self):
|
||||
|
|
@ -84,15 +84,15 @@ class SelectorTestCase(unittest.TestCase):
|
|||
headers = {'Content-Type': ['text/html; charset=utf-8']}
|
||||
response = HtmlResponse(url="http://example.com", headers=headers, body=html_utf8)
|
||||
x = Selector(response)
|
||||
self.assertEqual(x.xpath("//span[@id='blank']/text()").extract(),
|
||||
self.assertEqual(x.xpath("//span[@id='blank']/text()").getall(),
|
||||
[u'\xa3'])
|
||||
|
||||
def test_badly_encoded_body(self):
|
||||
# \xe9 alone isn't valid utf8 sequence
|
||||
r1 = TextResponse('http://www.example.com', \
|
||||
body=b'<html><p>an Jos\xe9 de</p><html>', \
|
||||
r1 = TextResponse('http://www.example.com',
|
||||
body=b'<html><p>an Jos\xe9 de</p><html>',
|
||||
encoding='utf-8')
|
||||
Selector(r1).xpath('//text()').extract()
|
||||
Selector(r1).xpath('//text()').getall()
|
||||
|
||||
def test_weakref_slots(self):
|
||||
"""Check that classes are using slots and are weak-referenceable"""
|
||||
|
|
|
|||
|
|
@ -147,10 +147,10 @@ class XMLFeedSpiderTest(SpiderTest):
|
|||
|
||||
def parse_node(self, response, selector):
|
||||
yield {
|
||||
'loc': selector.xpath('a:loc/text()').extract(),
|
||||
'updated': selector.xpath('b:updated/text()').extract(),
|
||||
'other': selector.xpath('other/@value').extract(),
|
||||
'custom': selector.xpath('other/@b:custom').extract(),
|
||||
'loc': selector.xpath('a:loc/text()').getall(),
|
||||
'updated': selector.xpath('b:updated/text()').getall(),
|
||||
'other': selector.xpath('other/@value').getall(),
|
||||
'custom': selector.xpath('other/@b:custom').getall(),
|
||||
}
|
||||
|
||||
for iterator in ('iternodes', 'xml'):
|
||||
|
|
|
|||
|
|
@ -7,7 +7,7 @@ from scrapy.utils.iterators import csviter, xmliter, _body_or_str, xmliter_lxml
|
|||
from scrapy.http import XmlResponse, TextResponse, Response
|
||||
from tests import get_testdata
|
||||
|
||||
FOOBAR_NL = u"foo" + os.linesep + u"bar"
|
||||
FOOBAR_NL = u"foo\nbar"
|
||||
|
||||
|
||||
class XmliterTestCase(unittest.TestCase):
|
||||
|
|
@ -30,10 +30,13 @@ class XmliterTestCase(unittest.TestCase):
|
|||
response = XmlResponse(url="http://example.com", body=body)
|
||||
attrs = []
|
||||
for x in self.xmliter(response, 'product'):
|
||||
attrs.append((x.xpath("@id").extract(), x.xpath("name/text()").extract(), x.xpath("./type/text()").extract()))
|
||||
attrs.append((
|
||||
x.attrib['id'],
|
||||
x.xpath("name/text()").getall(),
|
||||
x.xpath("./type/text()").getall()))
|
||||
|
||||
self.assertEqual(attrs,
|
||||
[(['001'], ['Name 1'], ['Type 1']), (['002'], ['Name 2'], ['Type 2'])])
|
||||
[('001', ['Name 1'], ['Type 1']), ('002', ['Name 2'], ['Type 2'])])
|
||||
|
||||
def test_xmliter_unusual_node(self):
|
||||
body = b"""<?xml version="1.0" encoding="UTF-8"?>
|
||||
|
|
@ -43,7 +46,7 @@ class XmliterTestCase(unittest.TestCase):
|
|||
</root>
|
||||
"""
|
||||
response = XmlResponse(url="http://example.com", body=body)
|
||||
nodenames = [e.xpath('name()').extract()
|
||||
nodenames = [e.xpath('name()').getall()
|
||||
for e in self.xmliter(response, 'matchme...')]
|
||||
self.assertEqual(nodenames, [['matchme...']])
|
||||
|
||||
|
|
@ -93,19 +96,19 @@ class XmliterTestCase(unittest.TestCase):
|
|||
|
||||
attrs = []
|
||||
for x in self.xmliter(r, u'þingflokkur'):
|
||||
attrs.append((x.xpath('@id').extract(),
|
||||
x.xpath(u'./skammstafanir/stuttskammstöfun/text()').extract(),
|
||||
x.xpath(u'./tímabil/fyrstaþing/text()').extract()))
|
||||
attrs.append((x.attrib['id'],
|
||||
x.xpath(u'./skammstafanir/stuttskammstöfun/text()').getall(),
|
||||
x.xpath(u'./tímabil/fyrstaþing/text()').getall()))
|
||||
|
||||
self.assertEqual(attrs,
|
||||
[([u'26'], [u'-'], [u'80']),
|
||||
([u'21'], [u'Ab'], [u'76']),
|
||||
([u'27'], [u'A'], [u'27'])])
|
||||
[(u'26', [u'-'], [u'80']),
|
||||
(u'21', [u'Ab'], [u'76']),
|
||||
(u'27', [u'A'], [u'27'])])
|
||||
|
||||
def test_xmliter_text(self):
|
||||
body = u"""<?xml version="1.0" encoding="UTF-8"?><products><product>one</product><product>two</product></products>"""
|
||||
|
||||
self.assertEqual([x.xpath("text()").extract() for x in self.xmliter(body, 'product')],
|
||||
self.assertEqual([x.xpath("text()").getall() for x in self.xmliter(body, 'product')],
|
||||
[[u'one'], [u'two']])
|
||||
|
||||
def test_xmliter_namespaces(self):
|
||||
|
|
@ -132,15 +135,15 @@ class XmliterTestCase(unittest.TestCase):
|
|||
|
||||
node = next(my_iter)
|
||||
node.register_namespace('g', 'http://base.google.com/ns/1.0')
|
||||
self.assertEqual(node.xpath('title/text()').extract(), ['Item 1'])
|
||||
self.assertEqual(node.xpath('description/text()').extract(), ['This is item 1'])
|
||||
self.assertEqual(node.xpath('link/text()').extract(), ['http://www.mydummycompany.com/items/1'])
|
||||
self.assertEqual(node.xpath('g:image_link/text()').extract(), ['http://www.mydummycompany.com/images/item1.jpg'])
|
||||
self.assertEqual(node.xpath('g:id/text()').extract(), ['ITEM_1'])
|
||||
self.assertEqual(node.xpath('g:price/text()').extract(), ['400'])
|
||||
self.assertEqual(node.xpath('image_link/text()').extract(), [])
|
||||
self.assertEqual(node.xpath('id/text()').extract(), [])
|
||||
self.assertEqual(node.xpath('price/text()').extract(), [])
|
||||
self.assertEqual(node.xpath('title/text()').getall(), ['Item 1'])
|
||||
self.assertEqual(node.xpath('description/text()').getall(), ['This is item 1'])
|
||||
self.assertEqual(node.xpath('link/text()').getall(), ['http://www.mydummycompany.com/items/1'])
|
||||
self.assertEqual(node.xpath('g:image_link/text()').getall(), ['http://www.mydummycompany.com/images/item1.jpg'])
|
||||
self.assertEqual(node.xpath('g:id/text()').getall(), ['ITEM_1'])
|
||||
self.assertEqual(node.xpath('g:price/text()').getall(), ['400'])
|
||||
self.assertEqual(node.xpath('image_link/text()').getall(), [])
|
||||
self.assertEqual(node.xpath('id/text()').getall(), [])
|
||||
self.assertEqual(node.xpath('price/text()').getall(), [])
|
||||
|
||||
def test_xmliter_exception(self):
|
||||
body = u"""<?xml version="1.0" encoding="UTF-8"?><products><product>one</product><product>two</product></products>"""
|
||||
|
|
@ -159,7 +162,7 @@ class XmliterTestCase(unittest.TestCase):
|
|||
body = b'<?xml version="1.0" encoding="ISO-8859-9"?>\n<xml>\n <item>Some Turkish Characters \xd6\xc7\xde\xdd\xd0\xdc \xfc\xf0\xfd\xfe\xe7\xf6</item>\n</xml>\n\n'
|
||||
response = XmlResponse('http://www.example.com', body=body)
|
||||
self.assertEqual(
|
||||
next(self.xmliter(response, 'item')).extract(),
|
||||
next(self.xmliter(response, 'item')).get(),
|
||||
u'<item>Some Turkish Characters \xd6\xc7\u015e\u0130\u011e\xdc \xfc\u011f\u0131\u015f\xe7\xf6</item>'
|
||||
)
|
||||
|
||||
|
|
@ -192,9 +195,9 @@ class LxmlXmliterTestCase(XmliterTestCase):
|
|||
|
||||
namespace_iter = self.xmliter(response, 'image_link', 'http://base.google.com/ns/1.0')
|
||||
node = next(namespace_iter)
|
||||
self.assertEqual(node.xpath('text()').extract(), ['http://www.mydummycompany.com/images/item1.jpg'])
|
||||
self.assertEqual(node.xpath('text()').getall(), ['http://www.mydummycompany.com/images/item1.jpg'])
|
||||
node = next(namespace_iter)
|
||||
self.assertEqual(node.xpath('text()').extract(), ['http://www.mydummycompany.com/images/item2.jpg'])
|
||||
self.assertEqual(node.xpath('text()').getall(), ['http://www.mydummycompany.com/images/item2.jpg'])
|
||||
|
||||
def test_xmliter_namespaces_prefix(self):
|
||||
body = b"""\
|
||||
|
|
@ -219,14 +222,14 @@ class LxmlXmliterTestCase(XmliterTestCase):
|
|||
my_iter = self.xmliter(response, 'table', 'http://www.w3.org/TR/html4/', 'h')
|
||||
|
||||
node = next(my_iter)
|
||||
self.assertEqual(len(node.xpath('h:tr/h:td').extract()), 2)
|
||||
self.assertEqual(node.xpath('h:tr/h:td[1]/text()').extract(), ['Apples'])
|
||||
self.assertEqual(node.xpath('h:tr/h:td[2]/text()').extract(), ['Bananas'])
|
||||
self.assertEqual(len(node.xpath('h:tr/h:td').getall()), 2)
|
||||
self.assertEqual(node.xpath('h:tr/h:td[1]/text()').getall(), ['Apples'])
|
||||
self.assertEqual(node.xpath('h:tr/h:td[2]/text()').getall(), ['Bananas'])
|
||||
|
||||
my_iter = self.xmliter(response, 'table', 'http://www.w3schools.com/furniture', 'f')
|
||||
|
||||
node = next(my_iter)
|
||||
self.assertEqual(node.xpath('f:name/text()').extract(), ['African Coffee Table'])
|
||||
self.assertEqual(node.xpath('f:name/text()').getall(), ['African Coffee Table'])
|
||||
|
||||
def test_xmliter_objtype_exception(self):
|
||||
i = self.xmliter(42, 'product')
|
||||
|
|
|
|||
|
|
@ -3,7 +3,9 @@ import os
|
|||
import unittest
|
||||
|
||||
from scrapy.item import Item, Field
|
||||
from scrapy.utils.misc import load_object, arg_to_iter, walk_modules
|
||||
from scrapy.utils.misc import arg_to_iter, create_instance, load_object, walk_modules
|
||||
|
||||
from tests import mock
|
||||
|
||||
__doctests__ = ['scrapy.utils.misc']
|
||||
|
||||
|
|
@ -74,5 +76,59 @@ class UtilsMiscTestCase(unittest.TestCase):
|
|||
self.assertEqual(list(arg_to_iter({'a':1})), [{'a': 1}])
|
||||
self.assertEqual(list(arg_to_iter(TestItem(name="john"))), [TestItem(name="john")])
|
||||
|
||||
def test_create_instance(self):
|
||||
settings = mock.MagicMock()
|
||||
crawler = mock.MagicMock(spec_set=['settings'])
|
||||
args = (True, 100.)
|
||||
kwargs = {'key': 'val'}
|
||||
|
||||
def _test_with_settings(mock, settings):
|
||||
create_instance(mock, settings, None, *args, **kwargs)
|
||||
if hasattr(mock, 'from_crawler'):
|
||||
self.assertEqual(mock.from_crawler.call_count, 0)
|
||||
if hasattr(mock, 'from_settings'):
|
||||
mock.from_settings.assert_called_once_with(settings, *args,
|
||||
**kwargs)
|
||||
self.assertEqual(mock.call_count, 0)
|
||||
else:
|
||||
mock.assert_called_once_with(*args, **kwargs)
|
||||
|
||||
def _test_with_crawler(mock, settings, crawler):
|
||||
create_instance(mock, settings, crawler, *args, **kwargs)
|
||||
if hasattr(mock, 'from_crawler'):
|
||||
mock.from_crawler.assert_called_once_with(crawler, *args,
|
||||
**kwargs)
|
||||
if hasattr(mock, 'from_settings'):
|
||||
self.assertEqual(mock.from_settings.call_count, 0)
|
||||
self.assertEqual(mock.call_count, 0)
|
||||
elif hasattr(mock, 'from_settings'):
|
||||
mock.from_settings.assert_called_once_with(settings, *args,
|
||||
**kwargs)
|
||||
self.assertEqual(mock.call_count, 0)
|
||||
else:
|
||||
mock.assert_called_once_with(*args, **kwargs)
|
||||
|
||||
# Check usage of correct constructor using four mocks:
|
||||
# 1. with no alternative constructors
|
||||
# 2. with from_settings() constructor
|
||||
# 3. with from_crawler() constructor
|
||||
# 4. with from_settings() and from_crawler() constructor
|
||||
spec_sets = ([], ['from_settings'], ['from_crawler'],
|
||||
['from_settings', 'from_crawler'])
|
||||
for specs in spec_sets:
|
||||
m = mock.MagicMock(spec_set=specs)
|
||||
_test_with_settings(m, settings)
|
||||
m.reset_mock()
|
||||
_test_with_crawler(m, settings, crawler)
|
||||
|
||||
# Check adoption of crawler settings
|
||||
m = mock.MagicMock(spec_set=['from_settings'])
|
||||
create_instance(m, None, crawler, *args, **kwargs)
|
||||
m.from_settings.assert_called_once_with(crawler.settings, *args,
|
||||
**kwargs)
|
||||
|
||||
with self.assertRaises(ValueError):
|
||||
create_instance(m, None, None)
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
|
|
@ -25,8 +25,12 @@ def inside_a_project():
|
|||
|
||||
class ProjectUtilsTest(unittest.TestCase):
|
||||
def test_data_path_outside_project(self):
|
||||
self.assertEqual('.scrapy/somepath', data_path('somepath'))
|
||||
self.assertEqual('/absolute/path', data_path('/absolute/path'))
|
||||
self.assertEqual(
|
||||
os.path.join('.scrapy', 'somepath'),
|
||||
data_path('somepath')
|
||||
)
|
||||
abspath = os.path.join(os.path.sep, 'absolute', 'path')
|
||||
self.assertEqual(abspath, data_path(abspath))
|
||||
|
||||
def test_data_path_inside_project(self):
|
||||
with inside_a_project() as proj_path:
|
||||
|
|
@ -35,4 +39,5 @@ class ProjectUtilsTest(unittest.TestCase):
|
|||
os.path.realpath(expected),
|
||||
os.path.realpath(data_path('somepath'))
|
||||
)
|
||||
self.assertEqual('/absolute/path', data_path('/absolute/path'))
|
||||
abspath = os.path.join(os.path.sep, 'absolute', 'path')
|
||||
self.assertEqual(abspath, data_path(abspath))
|
||||
|
|
|
|||