Merge pull request #3390 from scrapy/parsel-1.5

[MRG+1] update Scrapy to use parsel 1.5
This commit is contained in:
Daniel Graña 2018-09-18 13:07:41 -03:00 committed by GitHub
commit bafd174a9f
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
26 changed files with 808 additions and 548 deletions

1
.gitignore vendored
View File

@ -15,6 +15,7 @@ htmlcov/
.pytest_cache/
.coverage.*
.cache/
.pytest_cache/
# Windows
Thumbs.db

View File

@ -34,11 +34,11 @@ http://quotes.toscrape.com, following the pagination::
def parse(self, response):
for quote in response.css('div.quote'):
yield {
'text': quote.css('span.text::text').extract_first(),
'author': quote.xpath('span/small/text()').extract_first(),
'text': quote.css('span.text::text').get(),
'author': quote.xpath('span/small/text()').get(),
}
next_page = response.css('li.next a::attr("href")').extract_first()
next_page = response.css('li.next a::attr("href")').get()
if next_page is not None:
yield response.follow(next_page, self.parse)

View File

@ -254,7 +254,7 @@ data.
To extract the text from the title above, you can do::
>>> response.css('title::text').extract()
>>> response.css('title::text').getall()
['Quotes to Scrape']
There are two things to note here: one is that we've added ``::text`` to the
@ -262,32 +262,33 @@ CSS query, to mean we want to select only the text elements directly inside
``<title>`` element. If we don't specify ``::text``, we'd get the full title
element, including its tags::
>>> response.css('title').extract()
>>> response.css('title').getall()
['<title>Quotes to Scrape</title>']
The other thing is that the result of calling ``.extract()`` is a list, because
we're dealing with an instance of :class:`~scrapy.selector.SelectorList`. When
you know you just want the first result, as in this case, you can do::
The other thing is that the result of calling ``.getall()`` is a list: it is
possible that a selector returns more than one result, so we extract them all.
When you know you just want the first result, as in this case, you can do::
>>> response.css('title::text').extract_first()
>>> response.css('title::text').get()
'Quotes to Scrape'
As an alternative, you could've written::
>>> response.css('title::text')[0].extract()
>>> response.css('title::text')[0].get()
'Quotes to Scrape'
However, using ``.extract_first()`` avoids an ``IndexError`` and returns
``None`` when it doesn't find any element matching the selection.
However, using ``.get()`` directly on a :class:`~scrapy.selector.SelectorList`
instance avoids an ``IndexError`` and returns ``None`` when it doesn't
find any element matching the selection.
There's a lesson here: for most scraping code, you want it to be resilient to
errors due to things not being found on a page, so that even if some parts fail
to be scraped, you can at least get **some** data.
Besides the :meth:`~scrapy.selector.Selector.extract` and
:meth:`~scrapy.selector.SelectorList.extract_first` methods, you can also use
the :meth:`~scrapy.selector.Selector.re` method to extract using `regular
expressions`::
Besides the :meth:`~scrapy.selector.SelectorList.getall` and
:meth:`~scrapy.selector.SelectorList.get` methods, you can also use
the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular
expressions`_::
>>> response.css('title::text').re(r'Quotes.*')
['Quotes to Scrape']
@ -298,7 +299,8 @@ expressions`::
In order to find the proper CSS selectors to use, you might find useful opening
the response page from the shell in your web browser using ``view(response)``.
You can use your browser developer tools (see section about :ref:`topics-developer-tools`).
You can use your browser developer tools to inspect the HTML and come up
with a selector (see section about :ref:`topics-developer-tools`).
`Selector Gadget`_ is also a nice tool to quickly find CSS selector for
visually selected elements, which works in many browsers.
@ -314,7 +316,7 @@ Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions::
>>> response.xpath('//title')
[<Selector xpath='//title' data='<title>Quotes to Scrape</title>'>]
>>> response.xpath('//title/text()').extract_first()
>>> response.xpath('//title/text()').get()
'Quotes to Scrape'
XPath expressions are very powerful, and are the foundation of Scrapy
@ -383,17 +385,17 @@ variable, so that we can run our CSS selectors directly on a particular quote::
Now, let's extract ``title``, ``author`` and the ``tags`` from that quote
using the ``quote`` object we just created::
>>> title = quote.css("span.text::text").extract_first()
>>> title = quote.css("span.text::text").get()
>>> title
'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'
>>> author = quote.css("small.author::text").extract_first()
>>> author = quote.css("small.author::text").get()
>>> author
'Albert Einstein'
Given that the tags are a list of strings, we can use the ``.extract()`` method
Given that the tags are a list of strings, we can use the ``.getall()`` method
to get all of them::
>>> tags = quote.css("div.tags a.tag::text").extract()
>>> tags = quote.css("div.tags a.tag::text").getall()
>>> tags
['change', 'deep-thoughts', 'thinking', 'world']
@ -401,9 +403,9 @@ Having figured out how to extract each bit, we can now iterate over all the
quotes elements and put them together into a Python dictionary::
>>> for quote in response.css("div.quote"):
... text = quote.css("span.text::text").extract_first()
... author = quote.css("small.author::text").extract_first()
... tags = quote.css("div.tags a.tag::text").extract()
... text = quote.css("span.text::text").get()
... author = quote.css("small.author::text").get()
... tags = quote.css("div.tags a.tag::text").getall()
... print(dict(text=text, author=author, tags=tags))
{'tags': ['change', 'deep-thoughts', 'thinking', 'world'], 'author': 'Albert Einstein', 'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'}
{'tags': ['abilities', 'choices'], 'author': 'J.K. Rowling', 'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”'}
@ -434,9 +436,9 @@ in the callback, as you can see below::
def parse(self, response):
for quote in response.css('div.quote'):
yield {
'text': quote.css('span.text::text').extract_first(),
'author': quote.css('small.author::text').extract_first(),
'tags': quote.css('div.tags a.tag::text').extract(),
'text': quote.css('span.text::text').get(),
'author': quote.css('small.author::text').get(),
'tags': quote.css('div.tags a.tag::text').getall(),
}
If you run this spider, it will output the extracted data with the log::
@ -508,16 +510,22 @@ markup:
We can try extracting it in the shell::
>>> response.css('li.next a').extract_first()
>>> response.css('li.next a').get()
'<a href="/page/2/">Next <span aria-hidden="true">→</span></a>'
This gets the anchor element, but we want the attribute ``href``. For that,
Scrapy supports a CSS extension that let's you select the attribute contents,
like this::
>>> response.css('li.next a::attr(href)').extract_first()
>>> response.css('li.next a::attr(href)').get()
'/page/2/'
There is also an ``attrib`` property available
(see :ref:`selecting-attributes` for more)::
>>> response.css('li.next a').attrib['href']
'/page/2'
Let's see now our spider modified to recursively follow the link to the next
page, extracting data from it::
@ -533,12 +541,12 @@ page, extracting data from it::
def parse(self, response):
for quote in response.css('div.quote'):
yield {
'text': quote.css('span.text::text').extract_first(),
'author': quote.css('small.author::text').extract_first(),
'tags': quote.css('div.tags a.tag::text').extract(),
'text': quote.css('span.text::text').get(),
'author': quote.css('small.author::text').get(),
'tags': quote.css('div.tags a.tag::text').getall(),
}
next_page = response.css('li.next a::attr(href)').extract_first()
next_page = response.css('li.next a::attr(href)').get()
if next_page is not None:
next_page = response.urljoin(next_page)
yield scrapy.Request(next_page, callback=self.parse)
@ -584,12 +592,12 @@ As a shortcut for creating Request objects you can use
def parse(self, response):
for quote in response.css('div.quote'):
yield {
'text': quote.css('span.text::text').extract_first(),
'author': quote.css('span small::text').extract_first(),
'tags': quote.css('div.tags a.tag::text').extract(),
'text': quote.css('span.text::text').get(),
'author': quote.css('span small::text').get(),
'tags': quote.css('div.tags a.tag::text').getall(),
}
next_page = response.css('li.next a::attr(href)').extract_first()
next_page = response.css('li.next a::attr(href)').get()
if next_page is not None:
yield response.follow(next_page, callback=self.parse)
@ -641,7 +649,7 @@ this time for scraping author information::
def parse_author(self, response):
def extract_with_css(query):
return response.css(query).extract_first().strip()
return response.css(query).get(default='').strip()
yield {
'name': extract_with_css('h3.author-title::text'),
@ -710,11 +718,11 @@ with a specific tag, building the URL based on the argument::
def parse(self, response):
for quote in response.css('div.quote'):
yield {
'text': quote.css('span.text::text').extract_first(),
'author': quote.css('small.author::text').extract_first(),
'text': quote.css('span.text::text').get(),
'author': quote.css('small.author::text').get(),
}
next_page = response.css('li.next a::attr(href)').extract_first()
next_page = response.css('li.next a::attr(href)').get()
if next_page is not None:
yield response.follow(next_page, self.parse)
@ -738,4 +746,3 @@ modeling the scraped data. If you prefer to play with an example project, check
the :ref:`intro-examples` section.
.. _JSON: https://en.wikipedia.org/wiki/JSON
.. _dirbot: https://github.com/scrapy/dirbot

View File

@ -458,9 +458,9 @@ Usage example::
>>> STATUS DEPTH LEVEL 1 <<<
# Scraped Items ------------------------------------------------------------
[{'name': u'Example item',
'category': u'Furniture',
'length': u'12 cm'}]
[{'name': 'Example item',
'category': 'Furniture',
'length': '12 cm'}]
# Requests -----------------------------------------------------------------
[]

View File

@ -86,7 +86,7 @@ Creating items
::
>>> product = Product(name='Desktop PC', price=1000)
>>> print product
>>> print(product)
Product(name='Desktop PC', price=1000)
Getting field values
@ -161,11 +161,11 @@ Other common tasks
Copying items::
>>> product2 = Product(product)
>>> print product2
>>> print(product2)
Product(name='Desktop PC', price=1000)
>>> product3 = product2.copy()
>>> print product3
>>> print(product3)
Product(name='Desktop PC', price=1000)
Creating dicts from items::

View File

@ -84,7 +84,7 @@ So, for example, this won't work::
return scrapy.Request('http://www.example.com', callback=lambda r: self.other_callback(r, somearg))
def other_callback(self, response, somearg):
print "the argument passed is:", somearg
print("the argument passed is: %s" % somearg)
But this will::
@ -94,7 +94,7 @@ But this will::
def other_callback(self, response):
somearg = response.meta['somearg']
print "the argument passed is:", somearg
print("the argument passed is: %s" % somearg)
If you wish to log the requests that couldn't be serialized, you can set the
:setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page.

View File

@ -678,10 +678,10 @@ Here is a list of all built-in processors:
>>> from scrapy.loader.processors import Join
>>> proc = Join()
>>> proc(['one', 'two', 'three'])
u'one two three'
'one two three'
>>> proc = Join('<br>')
>>> proc(['one', 'two', 'three'])
u'one<br>two<br>three'
'one<br>two<br>three'
.. class:: Compose(\*functions, \**default_loader_context)
@ -744,9 +744,9 @@ Here is a list of all built-in processors:
... return None if x == 'world' else x
...
>>> from scrapy.loader.processors import MapCompose
>>> proc = MapCompose(filter_world, unicode.upper)
>>> proc([u'hello', u'world', u'this', u'is', u'scrapy'])
[u'HELLO, u'THIS', u'IS', u'SCRAPY']
>>> proc = MapCompose(filter_world, str.upper)
>>> proc(['hello', 'world', 'this', 'is', 'scrapy'])
['HELLO, 'THIS', 'IS', 'SCRAPY']
As with the Compose processor, functions can receive Loader contexts, and
constructor keyword arguments are used as default context values. See
@ -772,7 +772,7 @@ Here is a list of all built-in processors:
>>> import json
>>> proc_single_json_str = Compose(json.loads, SelectJmes("foo"))
>>> proc_single_json_str('{"foo": "bar"}')
u'bar'
'bar'
>>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo')))
>>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]')
[u'bar']
['bar']

File diff suppressed because it is too large Load Diff

View File

@ -871,7 +871,7 @@ LOG_STDOUT
Default: ``False``
If ``True``, all standard output (and error) of your process will be redirected
to the log. For example if you ``print 'hello'`` it will appear in the Scrapy
to the log. For example if you ``print('hello')`` it will appear in the Scrapy
log.
.. setting:: LOG_SHORT_NAMES

View File

@ -179,13 +179,13 @@ all start with the ``[s]`` prefix)::
After that, we can start playing with the objects::
>>> response.xpath('//title/text()').extract_first()
>>> response.xpath('//title/text()').get()
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
>>> fetch("https://reddit.com")
>>> response.xpath('//title/text()').extract()
['reddit: the front page of the internet']
>>> response.xpath('//title/text()').get()
'reddit: the front page of the internet'
>>> request = request.replace(method="POST")

View File

@ -229,11 +229,11 @@ Return multiple Requests and items from a single callback::
]
def parse(self, response):
for h3 in response.xpath('//h3').extract():
for h3 in response.xpath('//h3').getall():
yield {"title": h3}
for url in response.xpath('//a/@href').extract():
yield scrapy.Request(url, callback=self.parse)
for href in response.xpath('//a/@href').getall():
yield scrapy.Request(response.urljoin(href), self.parse)
Instead of :attr:`~.start_urls` you can use :meth:`~.start_requests` directly;
to give data more structure you can use :ref:`topics-items`::
@ -251,11 +251,11 @@ to give data more structure you can use :ref:`topics-items`::
yield scrapy.Request('http://www.example.com/3.html', self.parse)
def parse(self, response):
for h3 in response.xpath('//h3').extract():
for h3 in response.xpath('//h3').getall():
yield MyItem(title=h3)
for url in response.xpath('//a/@href').extract():
yield scrapy.Request(url, callback=self.parse)
for href in response.xpath('//a/@href').getall():
yield scrapy.Request(response.urljoin(href), self.parse)
.. _spiderargs:
@ -434,8 +434,8 @@ Let's now take a look at an example CrawlSpider with rules::
self.logger.info('Hi, this is an item page! %s', response.url)
item = scrapy.Item()
item['id'] = response.xpath('//td[@id="item_id"]/text()').re(r'ID: (\d+)')
item['name'] = response.xpath('//td[@id="item_name"]/text()').extract()
item['description'] = response.xpath('//td[@id="item_description"]/text()').extract()
item['name'] = response.xpath('//td[@id="item_name"]/text()').get()
item['description'] = response.xpath('//td[@id="item_description"]/text()').get()
return item
@ -545,12 +545,12 @@ These spiders are pretty easy to use, let's have a look at one example::
itertag = 'item'
def parse_node(self, response, node):
self.logger.info('Hi, this is a <%s> node!: %s', self.itertag, ''.join(node.extract()))
self.logger.info('Hi, this is a <%s> node!: %s', self.itertag, ''.join(node.getall()))
item = TestItem()
item['id'] = node.xpath('@id').extract()
item['name'] = node.xpath('name').extract()
item['description'] = node.xpath('description').extract()
item['id'] = node.xpath('@id').get()
item['name'] = node.xpath('name').get()
item['description'] = node.xpath('description').get()
return item
Basically what we did up there was to create a spider that downloads a feed from

View File

@ -6,5 +6,5 @@ queuelib
w3lib>=1.17.0
six>=1.5.2
PyDispatcher>=2.0.5
parsel>=1.4
parsel>=1.5
service_identity

View File

@ -6,5 +6,5 @@ queuelib>=1.1.1
w3lib>=1.17.0
six>=1.5.2
PyDispatcher>=2.0.5
parsel>=1.4
parsel>=1.5
service_identity

View File

@ -141,7 +141,7 @@ class SgmlLinkExtractor(FilteringLinkExtractor):
base_url = get_base_url(response)
body = u''.join(f
for x in self.restrict_xpaths
for f in response.xpath(x).extract()
for f in response.xpath(x).getall()
).encode(response.encoding, errors='xmlcharrefreplace')
else:
body = response.body

View File

@ -181,7 +181,7 @@ class ItemLoader(object):
def _get_xpathvalues(self, xpaths, **kw):
self._check_selector_method()
xpaths = arg_to_iter(xpaths)
return flatten(self.selector.xpath(xpath).extract() for xpath in xpaths)
return flatten(self.selector.xpath(xpath).getall() for xpath in xpaths)
def add_css(self, field_name, css, *processors, **kw):
values = self._get_cssvalues(css, **kw)
@ -198,6 +198,6 @@ class ItemLoader(object):
def _get_cssvalues(self, csss, **kw):
self._check_selector_method()
csss = arg_to_iter(csss)
return flatten(self.selector.css(css).extract() for css in csss)
return flatten(self.selector.css(css).getall() for css in csss)
XPathItemLoader = create_deprecated_class('XPathItemLoader', ItemLoader)

View File

@ -27,6 +27,10 @@ def _response_from_text(text, st):
class SelectorList(_ParselSelector.selectorlist_cls, object_ref):
"""
The :class:`SelectorList` class is a subclass of the builtin ``list``
class, which provides a few additional methods.
"""
@deprecated(use_instead='.extract()')
def extract_unquoted(self):
return [x.extract_unquoted() for x in self]
@ -41,6 +45,35 @@ class SelectorList(_ParselSelector.selectorlist_cls, object_ref):
class Selector(_ParselSelector, object_ref):
"""
An instance of :class:`Selector` is a wrapper over response to select
certain parts of its content.
``response`` is an :class:`~scrapy.http.HtmlResponse` or an
:class:`~scrapy.http.XmlResponse` object that will be used for selecting
and extracting data.
``text`` is a unicode string or utf-8 encoded text for cases when a
``response`` isn't available. Using ``text`` and ``response`` together is
undefined behavior.
``type`` defines the selector type, it can be ``"html"``, ``"xml"``
or ``None`` (default).
If ``type`` is ``None``, the selector automatically chooses the best type
based on ``response`` type (see below), or defaults to ``"html"`` in case it
is used together with ``text``.
If ``type`` is ``None`` and a ``response`` is passed, the selector type is
inferred from the response type as follows:
* ``"html"`` for :class:`~scrapy.http.HtmlResponse` type
* ``"xml"`` for :class:`~scrapy.http.XmlResponse` type
* ``"html"`` for anything else
Otherwise, if ``type`` is set, the selector type will be forced and no
detection will occur.
"""
__slots__ = ['response']
selectorlist_cls = SelectorList

View File

@ -14,8 +14,8 @@ class $classname(CrawlSpider):
)
def parse_item(self, response):
i = {}
#i['domain_id'] = response.xpath('//input[@id="sid"]/@value').extract()
#i['name'] = response.xpath('//div[@id="name"]').extract()
#i['description'] = response.xpath('//div[@id="description"]').extract()
return i
item = {}
#item['domain_id'] = response.xpath('//input[@id="sid"]/@value').get()
#item['name'] = response.xpath('//div[@id="name"]').get()
#item['description'] = response.xpath('//div[@id="description"]').get()
return item

View File

@ -10,8 +10,8 @@ class $classname(XMLFeedSpider):
itertag = 'item' # change it accordingly
def parse_node(self, response, selector):
i = {}
#i['url'] = selector.select('url').extract()
#i['name'] = selector.select('name').extract()
#i['description'] = selector.select('description').extract()
return i
item = {}
#item['url'] = selector.select('url').get()
#item['name'] = selector.select('name').get()
#item['description'] = selector.select('description').get()
return item

View File

@ -71,7 +71,7 @@ setup(
'pyOpenSSL',
'cssselect>=0.9',
'six>=1.5.2',
'parsel>=1.4',
'parsel>=1.5',
'PyDispatcher>=2.0.5',
'service_identity',
],

View File

@ -35,7 +35,7 @@ class ShellTest(ProcessTest, SiteTest, unittest.TestCase):
@defer.inlineCallbacks
def test_response_selector_html(self):
xpath = 'response.xpath("//p[@class=\'one\']/text()").extract()[0]'
xpath = 'response.xpath("//p[@class=\'one\']/text()").get()'
_, out, _ = yield self.execute([self.url('/html'), '-c', xpath])
self.assertEqual(out.strip(), b'Works')

View File

@ -336,11 +336,11 @@ class TextResponseTest(BaseResponseTest):
self.assertIs(response.selector.response, response)
self.assertEqual(
response.selector.xpath("//title/text()").extract(),
response.selector.xpath("//title/text()").getall(),
[u'Some page']
)
self.assertEqual(
response.selector.css("title::text").extract(),
response.selector.css("title::text").getall(),
[u'Some page']
)
self.assertEqual(
@ -353,12 +353,12 @@ class TextResponseTest(BaseResponseTest):
response = self.response_class("http://www.example.com", body=body)
self.assertEqual(
response.xpath("//title/text()").extract(),
response.selector.xpath("//title/text()").extract(),
response.xpath("//title/text()").getall(),
response.selector.xpath("//title/text()").getall(),
)
self.assertEqual(
response.css("title::text").extract(),
response.selector.css("title::text").extract(),
response.css("title::text").getall(),
response.selector.css("title::text").getall(),
)
def test_selector_shortcuts_kwargs(self):
@ -366,13 +366,13 @@ class TextResponseTest(BaseResponseTest):
response = self.response_class("http://www.example.com", body=body)
self.assertEqual(
response.xpath("normalize-space(//p[@class=$pclass])", pclass="content").extract(),
response.xpath("normalize-space(//p[@class=\"content\"])").extract(),
response.xpath("normalize-space(//p[@class=$pclass])", pclass="content").getall(),
response.xpath("normalize-space(//p[@class=\"content\"])").getall(),
)
self.assertEqual(
response.xpath("//title[count(following::p[@class=$pclass])=$pcount]/text()",
pclass="content", pcount=1).extract(),
response.xpath("//title[count(following::p[@class=\"content\"])=1]/text()").extract(),
pclass="content", pcount=1).getall(),
response.xpath("//title[count(following::p[@class=\"content\"])=1]/text()").getall(),
)
def test_urljoin_with_base_url(self):
@ -562,7 +562,7 @@ class XmlResponseTest(TextResponseTest):
self.assertIs(response.selector.response, response)
self.assertEqual(
response.selector.xpath("//elem/text()").extract(),
response.selector.xpath("//elem/text()").getall(),
[u'value']
)
@ -571,8 +571,8 @@ class XmlResponseTest(TextResponseTest):
response = self.response_class("http://www.example.com", body=body)
self.assertEqual(
response.xpath("//elem/text()").extract(),
response.selector.xpath("//elem/text()").extract(),
response.xpath("//elem/text()").getall(),
response.selector.xpath("//elem/text()").getall(),
)
def test_selector_shortcuts_kwargs(self):
@ -583,12 +583,12 @@ class XmlResponseTest(TextResponseTest):
response = self.response_class("http://www.example.com", body=body)
self.assertEqual(
response.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).extract(),
response.selector.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).extract(),
response.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).getall(),
response.selector.xpath("//s:elem/text()", namespaces={'s': 'http://scrapy.org'}).getall(),
)
response.selector.register_namespace('s2', 'http://scrapy.org')
self.assertEqual(
response.xpath("//s1:elem/text()", namespaces={'s1': 'http://scrapy.org'}).extract(),
response.selector.xpath("//s2:elem/text()").extract(),
response.xpath("//s1:elem/text()", namespaces={'s1': 'http://scrapy.org'}).getall(),
response.selector.xpath("//s2:elem/text()").getall(),
)

View File

@ -634,7 +634,7 @@ class SubselectorLoaderTest(unittest.TestCase):
nl = l.nested_xpath("//header")
nl.add_xpath('name', 'div/text()')
nl.add_css('name_div', '#id')
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').extract())
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').getall())
self.assertEqual(l.get_output_value('name'), [u'marta'])
self.assertEqual(l.get_output_value('name_div'), [u'<div id="id">marta</div>'])
@ -649,7 +649,7 @@ class SubselectorLoaderTest(unittest.TestCase):
nl = l.nested_css("header")
nl.add_xpath('name', 'div/text()')
nl.add_css('name_div', '#id')
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').extract())
nl.add_value('name_value', nl.selector.xpath('div[@id = "id"]/text()').getall())
self.assertEqual(l.get_output_value('name'), [u'marta'])
self.assertEqual(l.get_output_value('name_div'), [u'<div id="id">marta</div>'])

View File

@ -29,7 +29,7 @@ class MediaDownloadSpider(SimpleSpider):
for href in response.xpath('''
//table[thead/tr/th="Filename"]
/tbody//a/@href
''').extract()],
''').getall()],
}
yield item

View File

@ -20,17 +20,17 @@ class SelectorTestCase(unittest.TestCase):
for x in xl:
assert isinstance(x, Selector)
self.assertEqual(sel.xpath('//input').extract(),
[x.extract() for x in sel.xpath('//input')])
self.assertEqual(sel.xpath('//input').getall(),
[x.get() for x in sel.xpath('//input')])
self.assertEqual([x.extract() for x in sel.xpath("//input[@name='a']/@name")],
self.assertEqual([x.get() for x in sel.xpath("//input[@name='a']/@name")],
[u'a'])
self.assertEqual([x.extract() for x in sel.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
self.assertEqual([x.get() for x in sel.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
[u'12.0'])
self.assertEqual(sel.xpath("concat('xpath', 'rules')").extract(),
self.assertEqual(sel.xpath("concat('xpath', 'rules')").getall(),
[u'xpathrules'])
self.assertEqual([x.extract() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
self.assertEqual([x.get() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
[u'12'])
def test_root_base_url(self):
@ -60,12 +60,12 @@ class SelectorTestCase(unittest.TestCase):
text = b'<div><img src="a.jpg"><p>Hello</div>'
sel = Selector(XmlResponse('http://example.com', body=text, encoding='utf-8'))
self.assertEqual(sel.type, 'xml')
self.assertEqual(sel.xpath("//div").extract(),
self.assertEqual(sel.xpath("//div").getall(),
[u'<div><img src="a.jpg"><p>Hello</p></img></div>'])
sel = Selector(HtmlResponse('http://example.com', body=text, encoding='utf-8'))
self.assertEqual(sel.type, 'html')
self.assertEqual(sel.xpath("//div").extract(),
self.assertEqual(sel.xpath("//div").getall(),
[u'<div><img src="a.jpg"><p>Hello</p></div>'])
def test_http_header_encoding_precedence(self):
@ -84,15 +84,15 @@ class SelectorTestCase(unittest.TestCase):
headers = {'Content-Type': ['text/html; charset=utf-8']}
response = HtmlResponse(url="http://example.com", headers=headers, body=html_utf8)
x = Selector(response)
self.assertEqual(x.xpath("//span[@id='blank']/text()").extract(),
self.assertEqual(x.xpath("//span[@id='blank']/text()").getall(),
[u'\xa3'])
def test_badly_encoded_body(self):
# \xe9 alone isn't valid utf8 sequence
r1 = TextResponse('http://www.example.com', \
body=b'<html><p>an Jos\xe9 de</p><html>', \
r1 = TextResponse('http://www.example.com',
body=b'<html><p>an Jos\xe9 de</p><html>',
encoding='utf-8')
Selector(r1).xpath('//text()').extract()
Selector(r1).xpath('//text()').getall()
def test_weakref_slots(self):
"""Check that classes are using slots and are weak-referenceable"""

View File

@ -147,10 +147,10 @@ class XMLFeedSpiderTest(SpiderTest):
def parse_node(self, response, selector):
yield {
'loc': selector.xpath('a:loc/text()').extract(),
'updated': selector.xpath('b:updated/text()').extract(),
'other': selector.xpath('other/@value').extract(),
'custom': selector.xpath('other/@b:custom').extract(),
'loc': selector.xpath('a:loc/text()').getall(),
'updated': selector.xpath('b:updated/text()').getall(),
'other': selector.xpath('other/@value').getall(),
'custom': selector.xpath('other/@b:custom').getall(),
}
for iterator in ('iternodes', 'xml'):

View File

@ -30,10 +30,13 @@ class XmliterTestCase(unittest.TestCase):
response = XmlResponse(url="http://example.com", body=body)
attrs = []
for x in self.xmliter(response, 'product'):
attrs.append((x.xpath("@id").extract(), x.xpath("name/text()").extract(), x.xpath("./type/text()").extract()))
attrs.append((
x.attrib['id'],
x.xpath("name/text()").getall(),
x.xpath("./type/text()").getall()))
self.assertEqual(attrs,
[(['001'], ['Name 1'], ['Type 1']), (['002'], ['Name 2'], ['Type 2'])])
[('001', ['Name 1'], ['Type 1']), ('002', ['Name 2'], ['Type 2'])])
def test_xmliter_unusual_node(self):
body = b"""<?xml version="1.0" encoding="UTF-8"?>
@ -43,7 +46,7 @@ class XmliterTestCase(unittest.TestCase):
</root>
"""
response = XmlResponse(url="http://example.com", body=body)
nodenames = [e.xpath('name()').extract()
nodenames = [e.xpath('name()').getall()
for e in self.xmliter(response, 'matchme...')]
self.assertEqual(nodenames, [['matchme...']])
@ -93,19 +96,19 @@ class XmliterTestCase(unittest.TestCase):
attrs = []
for x in self.xmliter(r, u'þingflokkur'):
attrs.append((x.xpath('@id').extract(),
x.xpath(u'./skammstafanir/stuttskammstöfun/text()').extract(),
x.xpath(u'./tímabil/fyrstaþing/text()').extract()))
attrs.append((x.attrib['id'],
x.xpath(u'./skammstafanir/stuttskammstöfun/text()').getall(),
x.xpath(u'./tímabil/fyrstaþing/text()').getall()))
self.assertEqual(attrs,
[([u'26'], [u'-'], [u'80']),
([u'21'], [u'Ab'], [u'76']),
([u'27'], [u'A'], [u'27'])])
[(u'26', [u'-'], [u'80']),
(u'21', [u'Ab'], [u'76']),
(u'27', [u'A'], [u'27'])])
def test_xmliter_text(self):
body = u"""<?xml version="1.0" encoding="UTF-8"?><products><product>one</product><product>two</product></products>"""
self.assertEqual([x.xpath("text()").extract() for x in self.xmliter(body, 'product')],
self.assertEqual([x.xpath("text()").getall() for x in self.xmliter(body, 'product')],
[[u'one'], [u'two']])
def test_xmliter_namespaces(self):
@ -132,15 +135,15 @@ class XmliterTestCase(unittest.TestCase):
node = next(my_iter)
node.register_namespace('g', 'http://base.google.com/ns/1.0')
self.assertEqual(node.xpath('title/text()').extract(), ['Item 1'])
self.assertEqual(node.xpath('description/text()').extract(), ['This is item 1'])
self.assertEqual(node.xpath('link/text()').extract(), ['http://www.mydummycompany.com/items/1'])
self.assertEqual(node.xpath('g:image_link/text()').extract(), ['http://www.mydummycompany.com/images/item1.jpg'])
self.assertEqual(node.xpath('g:id/text()').extract(), ['ITEM_1'])
self.assertEqual(node.xpath('g:price/text()').extract(), ['400'])
self.assertEqual(node.xpath('image_link/text()').extract(), [])
self.assertEqual(node.xpath('id/text()').extract(), [])
self.assertEqual(node.xpath('price/text()').extract(), [])
self.assertEqual(node.xpath('title/text()').getall(), ['Item 1'])
self.assertEqual(node.xpath('description/text()').getall(), ['This is item 1'])
self.assertEqual(node.xpath('link/text()').getall(), ['http://www.mydummycompany.com/items/1'])
self.assertEqual(node.xpath('g:image_link/text()').getall(), ['http://www.mydummycompany.com/images/item1.jpg'])
self.assertEqual(node.xpath('g:id/text()').getall(), ['ITEM_1'])
self.assertEqual(node.xpath('g:price/text()').getall(), ['400'])
self.assertEqual(node.xpath('image_link/text()').getall(), [])
self.assertEqual(node.xpath('id/text()').getall(), [])
self.assertEqual(node.xpath('price/text()').getall(), [])
def test_xmliter_exception(self):
body = u"""<?xml version="1.0" encoding="UTF-8"?><products><product>one</product><product>two</product></products>"""
@ -159,7 +162,7 @@ class XmliterTestCase(unittest.TestCase):
body = b'<?xml version="1.0" encoding="ISO-8859-9"?>\n<xml>\n <item>Some Turkish Characters \xd6\xc7\xde\xdd\xd0\xdc \xfc\xf0\xfd\xfe\xe7\xf6</item>\n</xml>\n\n'
response = XmlResponse('http://www.example.com', body=body)
self.assertEqual(
next(self.xmliter(response, 'item')).extract(),
next(self.xmliter(response, 'item')).get(),
u'<item>Some Turkish Characters \xd6\xc7\u015e\u0130\u011e\xdc \xfc\u011f\u0131\u015f\xe7\xf6</item>'
)
@ -192,9 +195,9 @@ class LxmlXmliterTestCase(XmliterTestCase):
namespace_iter = self.xmliter(response, 'image_link', 'http://base.google.com/ns/1.0')
node = next(namespace_iter)
self.assertEqual(node.xpath('text()').extract(), ['http://www.mydummycompany.com/images/item1.jpg'])
self.assertEqual(node.xpath('text()').getall(), ['http://www.mydummycompany.com/images/item1.jpg'])
node = next(namespace_iter)
self.assertEqual(node.xpath('text()').extract(), ['http://www.mydummycompany.com/images/item2.jpg'])
self.assertEqual(node.xpath('text()').getall(), ['http://www.mydummycompany.com/images/item2.jpg'])
def test_xmliter_namespaces_prefix(self):
body = b"""\
@ -219,14 +222,14 @@ class LxmlXmliterTestCase(XmliterTestCase):
my_iter = self.xmliter(response, 'table', 'http://www.w3.org/TR/html4/', 'h')
node = next(my_iter)
self.assertEqual(len(node.xpath('h:tr/h:td').extract()), 2)
self.assertEqual(node.xpath('h:tr/h:td[1]/text()').extract(), ['Apples'])
self.assertEqual(node.xpath('h:tr/h:td[2]/text()').extract(), ['Bananas'])
self.assertEqual(len(node.xpath('h:tr/h:td').getall()), 2)
self.assertEqual(node.xpath('h:tr/h:td[1]/text()').getall(), ['Apples'])
self.assertEqual(node.xpath('h:tr/h:td[2]/text()').getall(), ['Bananas'])
my_iter = self.xmliter(response, 'table', 'http://www.w3schools.com/furniture', 'f')
node = next(my_iter)
self.assertEqual(node.xpath('f:name/text()').extract(), ['African Coffee Table'])
self.assertEqual(node.xpath('f:name/text()').getall(), ['African Coffee Table'])
def test_xmliter_objtype_exception(self):
i = self.xmliter(42, 'product')