mirror of https://github.com/scrapy/scrapy.git
port tests to new Selector class
This commit is contained in:
parent
e4d6e2eb31
commit
4e94b38396
|
|
@ -12,6 +12,19 @@ class XPathSelector(Selector):
|
|||
__slots__ = ()
|
||||
default_contenttype = 'html'
|
||||
|
||||
def __init__(self, *a, **kw):
|
||||
import warnings
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
warnings.warn('%s is deprecated, instanciate scrapy.selector.Selector '
|
||||
'instead' % type(self).__name__,
|
||||
category=ScrapyDeprecationWarning, stacklevel=1)
|
||||
super(XPathSelector, self).__init__(*a, **kw)
|
||||
|
||||
def css(self, *a, **kw):
|
||||
raise RuntimeError('.css() method not available for %s, '
|
||||
'instanciate scrapy.selector.Selector '
|
||||
'instead' % type(self).__name__)
|
||||
|
||||
|
||||
class XmlXPathSelector(XPathSelector):
|
||||
__slots__ = ()
|
||||
|
|
@ -28,7 +41,7 @@ class XPathSelectorList(SelectorList):
|
|||
def __init__(self, *a, **kw):
|
||||
import warnings
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
warnings.warn('XPathSelectorList is deprecated, use '
|
||||
warnings.warn('XPathSelectorList is deprecated, instanciate '
|
||||
'scrapy.selector.SelectorList instead',
|
||||
category=ScrapyDeprecationWarning, stacklevel=1)
|
||||
super(XPathSelectorList, self).__init__(*a, **kw)
|
||||
|
|
|
|||
|
|
@ -1,81 +1,85 @@
|
|||
"""
|
||||
Selectors tests, common for all backends
|
||||
"""
|
||||
|
||||
import re
|
||||
import warnings
|
||||
import weakref
|
||||
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import TextResponse, HtmlResponse, XmlResponse
|
||||
from scrapy.selector import XmlXPathSelector, HtmlXPathSelector, \
|
||||
XPathSelector
|
||||
from scrapy.selector import Selector
|
||||
from scrapy.selector.lxmlsel import XmlXPathSelector, HtmlXPathSelector, XPathSelector
|
||||
|
||||
|
||||
class XPathSelectorTestCase(unittest.TestCase):
|
||||
class SelectorTestCase(unittest.TestCase):
|
||||
|
||||
xs_cls = XPathSelector
|
||||
hxs_cls = HtmlXPathSelector
|
||||
xxs_cls = XmlXPathSelector
|
||||
sscls = Selector
|
||||
|
||||
def test_selector_simple(self):
|
||||
def test_simple_selection(self):
|
||||
"""Simple selector tests"""
|
||||
body = "<p><input name='a'value='1'/><input name='b'value='2'/></p>"
|
||||
response = TextResponse(url="http://example.com", body=body)
|
||||
xpath = self.hxs_cls(response)
|
||||
ss = self.sscls(response)
|
||||
|
||||
xl = xpath.select('//input')
|
||||
xl = ss.xpath('//input')
|
||||
self.assertEqual(2, len(xl))
|
||||
for x in xl:
|
||||
assert isinstance(x, self.hxs_cls)
|
||||
assert isinstance(x, self.sscls)
|
||||
|
||||
self.assertEqual(xpath.select('//input').extract(),
|
||||
[x.extract() for x in xpath.select('//input')])
|
||||
self.assertEqual(ss.xpath('//input').extract(),
|
||||
[x.extract() for x in ss.xpath('//input')])
|
||||
|
||||
self.assertEqual([x.extract() for x in xpath.select("//input[@name='a']/@name")],
|
||||
self.assertEqual([x.extract() for x in ss.xpath("//input[@name='a']/@name")],
|
||||
[u'a'])
|
||||
self.assertEqual([x.extract() for x in xpath.select("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
|
||||
self.assertEqual([x.extract() for x in ss.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
|
||||
[u'12.0'])
|
||||
|
||||
self.assertEqual(xpath.select("concat('xpath', 'rules')").extract(),
|
||||
self.assertEqual(ss.xpath("concat('xpath', 'rules')").extract(),
|
||||
[u'xpathrules'])
|
||||
self.assertEqual([x.extract() for x in xpath.select("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
|
||||
self.assertEqual([x.extract() for x in ss.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
|
||||
[u'12'])
|
||||
|
||||
def test_selector_unicode_query(self):
|
||||
def test_select_unicode_query(self):
|
||||
body = u"<p><input name='\xa9' value='1'/></p>"
|
||||
response = TextResponse(url="http://example.com", body=body, encoding='utf8')
|
||||
xpath = self.hxs_cls(response)
|
||||
self.assertEqual(xpath.select(u'//input[@name="\xa9"]/@value').extract(), [u'1'])
|
||||
ss = self.sscls(response)
|
||||
self.assertEqual(ss.xpath(u'//input[@name="\xa9"]/@value').extract(), [u'1'])
|
||||
|
||||
def test_selector_same_type(self):
|
||||
"""Test XPathSelector returning the same type in x() method"""
|
||||
def test_list_elements_type(self):
|
||||
"""Test Selector returning the same type in selection methods"""
|
||||
text = '<p>test<p>'
|
||||
assert isinstance(self.xxs_cls(text=text).select("//p")[0],
|
||||
self.xxs_cls)
|
||||
assert isinstance(self.hxs_cls(text=text).select("//p")[0],
|
||||
self.hxs_cls)
|
||||
assert isinstance(self.sscls(text=text).xpath("//p")[0], self.sscls)
|
||||
assert isinstance(self.sscls(text=text).css("p")[0], self.sscls)
|
||||
|
||||
def test_selector_boolean_result(self):
|
||||
def test_boolean_result(self):
|
||||
body = "<p><input name='a'value='1'/><input name='b'value='2'/></p>"
|
||||
response = TextResponse(url="http://example.com", body=body)
|
||||
xs = self.hxs_cls(response)
|
||||
self.assertEquals(xs.select("//input[@name='a']/@name='a'").extract(), [u'1'])
|
||||
self.assertEquals(xs.select("//input[@name='a']/@name='n'").extract(), [u'0'])
|
||||
|
||||
def test_selector_xml_html(self):
|
||||
"""Test that XML and HTML XPathSelector's behave differently"""
|
||||
xs = self.sscls(response)
|
||||
self.assertEquals(xs.xpath("//input[@name='a']/@name='a'").extract(), [u'1'])
|
||||
self.assertEquals(xs.xpath("//input[@name='a']/@name='n'").extract(), [u'0'])
|
||||
|
||||
def test_differences_parsing_xml_vs_html(self):
|
||||
"""Test that XML and HTML Selector's behave differently"""
|
||||
# some text which is parsed differently by XML and HTML flavors
|
||||
text = '<div><img src="a.jpg"><p>Hello</div>'
|
||||
|
||||
self.assertEqual(self.xxs_cls(text=text).select("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></img></div>'])
|
||||
|
||||
self.assertEqual(self.hxs_cls(text=text).select("//div").extract(),
|
||||
hs = self.sscls(text=text, contenttype='html')
|
||||
self.assertEqual(hs.xpath("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></div>'])
|
||||
|
||||
def test_selector_nested(self):
|
||||
xs = self.sscls(text=text, contenttype='xml')
|
||||
self.assertEqual(xs.xpath("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></img></div>'])
|
||||
|
||||
def test_flavor_detection(self):
|
||||
text = '<div><img src="a.jpg"><p>Hello</div>'
|
||||
ss = self.sscls(XmlResponse('http://example.com', body=text))
|
||||
self.assertEqual(ss.contenttype, 'xml')
|
||||
self.assertEqual(ss.xpath("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></img></div>'])
|
||||
|
||||
ss = self.sscls(HtmlResponse('http://example.com', body=text))
|
||||
self.assertEqual(ss.contenttype, 'html')
|
||||
self.assertEqual(ss.xpath("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></div>'])
|
||||
|
||||
def test_nested_selectors(self):
|
||||
"""Nested selector tests"""
|
||||
body = """<body>
|
||||
<div class='one'>
|
||||
|
|
@ -91,24 +95,30 @@ class XPathSelectorTestCase(unittest.TestCase):
|
|||
</body>"""
|
||||
|
||||
response = HtmlResponse(url="http://example.com", body=body)
|
||||
x = self.hxs_cls(response)
|
||||
|
||||
divtwo = x.select('//div[@class="two"]')
|
||||
self.assertEqual(map(unicode.strip, divtwo.select("//li").extract()),
|
||||
x = self.sscls(response)
|
||||
divtwo = x.xpath('//div[@class="two"]')
|
||||
self.assertEqual(divtwo.xpath("//li").extract(),
|
||||
["<li>one</li>", "<li>two</li>", "<li>four</li>", "<li>five</li>", "<li>six</li>"])
|
||||
self.assertEqual(map(unicode.strip, divtwo.select("./ul/li").extract()),
|
||||
self.assertEqual(divtwo.xpath("./ul/li").extract(),
|
||||
["<li>four</li>", "<li>five</li>", "<li>six</li>"])
|
||||
self.assertEqual(map(unicode.strip, divtwo.select(".//li").extract()),
|
||||
self.assertEqual(divtwo.xpath(".//li").extract(),
|
||||
["<li>four</li>", "<li>five</li>", "<li>six</li>"])
|
||||
self.assertEqual(divtwo.select("./li").extract(),
|
||||
[])
|
||||
self.assertEqual(divtwo.xpath("./li").extract(), [])
|
||||
|
||||
def test_mixed_nested_selectors(self):
|
||||
body = '''<body>
|
||||
<div id=1>not<span>me</span></div>
|
||||
<div class="dos"><p>text</p><a href='#'>foo</a></div>
|
||||
</body>'''
|
||||
ss = self.sscls(text=body)
|
||||
self.assertEqual(ss.xpath('//div[@id="1"]').css('span::text').extract(), [u'me'])
|
||||
self.assertEqual(ss.css('#1').xpath('./span/text()').extract(), [u'me'])
|
||||
|
||||
def test_dont_strip(self):
|
||||
hxs = self.hxs_cls(text='<div>fff: <a href="#">zzz</a></div>')
|
||||
self.assertEqual(hxs.select("//text()").extract(),
|
||||
[u'fff: ', u'zzz'])
|
||||
hxs = self.sscls(text='<div>fff: <a href="#">zzz</a></div>')
|
||||
self.assertEqual(hxs.xpath("//text()").extract(), [u'fff: ', u'zzz'])
|
||||
|
||||
def test_selector_namespaces_simple(self):
|
||||
def test_namespaces_simple(self):
|
||||
body = """
|
||||
<test xmlns:somens="http://scrapy.org">
|
||||
<somens:a id="foo">take this</a>
|
||||
|
|
@ -117,13 +127,13 @@ class XPathSelectorTestCase(unittest.TestCase):
|
|||
"""
|
||||
|
||||
response = XmlResponse(url="http://example.com", body=body)
|
||||
x = self.xxs_cls(response)
|
||||
x = self.sscls(response)
|
||||
|
||||
x.register_namespace("somens", "http://scrapy.org")
|
||||
self.assertEqual(x.select("//somens:a/text()").extract(),
|
||||
self.assertEqual(x.xpath("//somens:a/text()").extract(),
|
||||
[u'take this'])
|
||||
|
||||
def test_selector_namespaces_multiple(self):
|
||||
def test_namespaces_multiple(self):
|
||||
body = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<BrowseNode xmlns="http://webservices.amazon.com/AWSECommerceService/2005-10-05"
|
||||
xmlns:b="http://somens.com"
|
||||
|
|
@ -134,19 +144,18 @@ class XPathSelectorTestCase(unittest.TestCase):
|
|||
</BrowseNode>
|
||||
"""
|
||||
response = XmlResponse(url="http://example.com", body=body)
|
||||
x = self.xxs_cls(response)
|
||||
|
||||
x = self.sscls(response)
|
||||
x.register_namespace("xmlns", "http://webservices.amazon.com/AWSECommerceService/2005-10-05")
|
||||
x.register_namespace("p", "http://www.scrapy.org/product")
|
||||
x.register_namespace("b", "http://somens.com")
|
||||
self.assertEqual(len(x.select("//xmlns:TestTag")), 1)
|
||||
self.assertEqual(x.select("//b:Operation/text()").extract()[0], 'hello')
|
||||
self.assertEqual(x.select("//xmlns:TestTag/@b:att").extract()[0], 'value')
|
||||
self.assertEqual(x.select("//p:SecondTestTag/xmlns:price/text()").extract()[0], '90')
|
||||
self.assertEqual(x.select("//p:SecondTestTag").select("./xmlns:price/text()")[0].extract(), '90')
|
||||
self.assertEqual(x.select("//p:SecondTestTag/xmlns:material/text()").extract()[0], 'iron')
|
||||
self.assertEqual(len(x.xpath("//xmlns:TestTag")), 1)
|
||||
self.assertEqual(x.xpath("//b:Operation/text()").extract()[0], 'hello')
|
||||
self.assertEqual(x.xpath("//xmlns:TestTag/@b:att").extract()[0], 'value')
|
||||
self.assertEqual(x.xpath("//p:SecondTestTag/xmlns:price/text()").extract()[0], '90')
|
||||
self.assertEqual(x.xpath("//p:SecondTestTag").xpath("./xmlns:price/text()")[0].extract(), '90')
|
||||
self.assertEqual(x.xpath("//p:SecondTestTag/xmlns:material/text()").extract()[0], 'iron')
|
||||
|
||||
def test_selector_re(self):
|
||||
def test_re(self):
|
||||
body = """<div>Name: Mary
|
||||
<ul>
|
||||
<li>Name: John</li>
|
||||
|
|
@ -155,44 +164,35 @@ class XPathSelectorTestCase(unittest.TestCase):
|
|||
<li>Age: 20</li>
|
||||
</ul>
|
||||
Age: 20
|
||||
</div>
|
||||
|
||||
"""
|
||||
</div>"""
|
||||
response = HtmlResponse(url="http://example.com", body=body)
|
||||
x = self.hxs_cls(response)
|
||||
x = self.sscls(response)
|
||||
|
||||
name_re = re.compile("Name: (\w+)")
|
||||
self.assertEqual(x.select("//ul/li").re(name_re),
|
||||
self.assertEqual(x.xpath("//ul/li").re(name_re),
|
||||
["John", "Paul"])
|
||||
self.assertEqual(x.select("//ul/li").re("Age: (\d+)"),
|
||||
self.assertEqual(x.xpath("//ul/li").re("Age: (\d+)"),
|
||||
["10", "20"])
|
||||
|
||||
def test_selector_re_intl(self):
|
||||
def test_re_intl(self):
|
||||
body = """<div>Evento: cumplea\xc3\xb1os</div>"""
|
||||
response = HtmlResponse(url="http://example.com", body=body, encoding='utf-8')
|
||||
x = self.hxs_cls(response)
|
||||
self.assertEqual(x.select("//div").re("Evento: (\w+)"), [u'cumplea\xf1os'])
|
||||
x = self.sscls(response)
|
||||
self.assertEqual(x.xpath("//div").re("Evento: (\w+)"), [u'cumplea\xf1os'])
|
||||
|
||||
def test_selector_over_text(self):
|
||||
hxs = self.hxs_cls(text='<root>lala</root>')
|
||||
self.assertEqual(hxs.extract(),
|
||||
u'<html><body><root>lala</root></body></html>')
|
||||
hs = self.sscls(text='<root>lala</root>')
|
||||
self.assertEqual(hs.extract(), u'<html><body><root>lala</root></body></html>')
|
||||
xs = self.sscls(text='<root>lala</root>', contenttype='xml')
|
||||
self.assertEqual(xs.extract(), u'<root>lala</root>')
|
||||
self.assertEqual(xs.xpath('.').extract(), [u'<root>lala</root>'])
|
||||
|
||||
xxs = self.xxs_cls(text='<root>lala</root>')
|
||||
self.assertEqual(xxs.extract(),
|
||||
u'<root>lala</root>')
|
||||
|
||||
xxs = self.xxs_cls(text='<root>lala</root>')
|
||||
self.assertEqual(xxs.select('.').extract(),
|
||||
[u'<root>lala</root>'])
|
||||
|
||||
|
||||
def test_selector_invalid_xpath(self):
|
||||
def test_invalid_xpath(self):
|
||||
response = XmlResponse(url="http://example.com", body="<html></html>")
|
||||
x = self.hxs_cls(response)
|
||||
x = self.sscls(response)
|
||||
xpath = "//test[@foo='bar]"
|
||||
try:
|
||||
x.select(xpath)
|
||||
x.xpath(xpath)
|
||||
except ValueError, e:
|
||||
assert xpath in str(e), "Exception message does not contain invalid xpath"
|
||||
except Exception:
|
||||
|
|
@ -215,64 +215,121 @@ class XPathSelectorTestCase(unittest.TestCase):
|
|||
|
||||
headers = {'Content-Type': ['text/html; charset=utf-8']}
|
||||
response = HtmlResponse(url="http://example.com", headers=headers, body=html_utf8)
|
||||
x = self.hxs_cls(response)
|
||||
self.assertEquals(x.select("//span[@id='blank']/text()").extract(),
|
||||
x = self.sscls(response)
|
||||
self.assertEquals(x.xpath("//span[@id='blank']/text()").extract(),
|
||||
[u'\xa3'])
|
||||
|
||||
def test_empty_bodies(self):
|
||||
# shouldn't raise errors
|
||||
r1 = TextResponse('http://www.example.com', body='')
|
||||
self.hxs_cls(r1).select('//text()').extract()
|
||||
self.xxs_cls(r1).select('//text()').extract()
|
||||
self.sscls(r1).xpath('//text()').extract()
|
||||
|
||||
def test_null_bytes(self):
|
||||
# shouldn't raise errors
|
||||
r1 = TextResponse('http://www.example.com', \
|
||||
body='<root>pre\x00post</root>', \
|
||||
encoding='utf-8')
|
||||
self.hxs_cls(r1).select('//text()').extract()
|
||||
self.xxs_cls(r1).select('//text()').extract()
|
||||
self.sscls(r1).xpath('//text()').extract()
|
||||
|
||||
def test_badly_encoded_body(self):
|
||||
# \xe9 alone isn't valid utf8 sequence
|
||||
r1 = TextResponse('http://www.example.com', \
|
||||
body='<html><p>an Jos\xe9 de</p><html>', \
|
||||
encoding='utf-8')
|
||||
self.hxs_cls(r1).select('//text()').extract()
|
||||
self.xxs_cls(r1).select('//text()').extract()
|
||||
self.sscls(r1).xpath('//text()').extract()
|
||||
|
||||
def test_select_on_unevaluable_nodes(self):
|
||||
r = self.hxs_cls(text=u'<span class="big">some text</span>')
|
||||
r = self.sscls(text=u'<span class="big">some text</span>')
|
||||
# Text node
|
||||
x1 = r.select('//text()')
|
||||
x1 = r.xpath('//text()')
|
||||
self.assertEquals(x1.extract(), [u'some text'])
|
||||
self.assertEquals(x1.select('.//b').extract(), [])
|
||||
self.assertEquals(x1.xpath('.//b').extract(), [])
|
||||
# Tag attribute
|
||||
x1 = r.select('//span/@class')
|
||||
x1 = r.xpath('//span/@class')
|
||||
self.assertEquals(x1.extract(), [u'big'])
|
||||
self.assertEquals(x1.select('.//text()').extract(), [])
|
||||
self.assertEquals(x1.xpath('.//text()').extract(), [])
|
||||
|
||||
def test_select_on_text_nodes(self):
|
||||
r = self.hxs_cls(text=u'<div><b>Options:</b>opt1</div><div><b>Other</b>opt2</div>')
|
||||
x1 = r.select("//div/descendant::text()[preceding-sibling::b[contains(text(), 'Options')]]")
|
||||
r = self.sscls(text=u'<div><b>Options:</b>opt1</div><div><b>Other</b>opt2</div>')
|
||||
x1 = r.xpath("//div/descendant::text()[preceding-sibling::b[contains(text(), 'Options')]]")
|
||||
self.assertEquals(x1.extract(), [u'opt1'])
|
||||
|
||||
x1 = r.select("//div/descendant::text()/preceding-sibling::b[contains(text(), 'Options')]")
|
||||
x1 = r.xpath("//div/descendant::text()/preceding-sibling::b[contains(text(), 'Options')]")
|
||||
self.assertEquals(x1.extract(), [u'<b>Options:</b>'])
|
||||
|
||||
def test_nested_select_on_text_nodes(self):
|
||||
# FIXME: does not work with lxml backend [upstream]
|
||||
r = self.hxs_cls(text=u'<div><b>Options:</b>opt1</div><div><b>Other</b>opt2</div>')
|
||||
x1 = r.select("//div/descendant::text()")
|
||||
x2 = x1.select("./preceding-sibling::b[contains(text(), 'Options')]")
|
||||
|
||||
r = self.sscls(text=u'<div><b>Options:</b>opt1</div><div><b>Other</b>opt2</div>')
|
||||
x1 = r.xpath("//div/descendant::text()")
|
||||
x2 = x1.xpath("./preceding-sibling::b[contains(text(), 'Options')]")
|
||||
self.assertEquals(x2.extract(), [u'<b>Options:</b>'])
|
||||
test_nested_select_on_text_nodes.skip = True
|
||||
test_nested_select_on_text_nodes.skip = "Text nodes lost parent node reference in lxml"
|
||||
|
||||
def test_weakref_slots(self):
|
||||
"""Check that classes are using slots and are weak-referenceable"""
|
||||
for cls in [self.xs_cls, self.hxs_cls, self.xxs_cls]:
|
||||
x = cls()
|
||||
weakref.ref(x)
|
||||
assert not hasattr(x, '__dict__'), "%s does not use __slots__" % \
|
||||
x.__class__.__name__
|
||||
x = self.sscls()
|
||||
weakref.ref(x)
|
||||
assert not hasattr(x, '__dict__'), "%s does not use __slots__" % \
|
||||
x.__class__.__name__
|
||||
|
||||
def test_remove_namespaces(self):
|
||||
xml = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<feed xmlns="http://www.w3.org/2005/Atom" xml:lang="en-US" xmlns:media="http://search.yahoo.com/mrss/">
|
||||
<link type="text/html">
|
||||
<link type="application/atom+xml">
|
||||
</feed>
|
||||
"""
|
||||
xxs = self.sscls(XmlResponse("http://example.com/feed.atom", body=xml))
|
||||
self.assertEqual(len(xxs.xpath("//link")), 0)
|
||||
xxs.remove_namespaces()
|
||||
self.assertEqual(len(xxs.xpath("//link")), 2)
|
||||
|
||||
def test_remove_attributes_namespaces(self):
|
||||
xml = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<feed xmlns:atom="http://www.w3.org/2005/Atom" xml:lang="en-US" xmlns:media="http://search.yahoo.com/mrss/">
|
||||
<link atom:type="text/html">
|
||||
<link atom:type="application/atom+xml">
|
||||
</feed>
|
||||
"""
|
||||
xxs = self.sscls(XmlResponse("http://example.com/feed.atom", body=xml))
|
||||
self.assertEqual(len(xxs.xpath("//link/@type")), 0)
|
||||
xxs.remove_namespaces()
|
||||
self.assertEqual(len(xxs.xpath("//link/@type")), 2)
|
||||
|
||||
|
||||
class DeprecatedXpathSelectorTest(unittest.TestCase):
|
||||
|
||||
text = '<div><img src="a.jpg"><p>Hello</div>'
|
||||
|
||||
def test_warnings(self):
|
||||
for cls in XPathSelector, HtmlXPathSelector, XPathSelector:
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
warnings.simplefilter('always')
|
||||
hs = cls(text=self.text)
|
||||
assert len(w) == 1, w
|
||||
assert issubclass(w[0].category, ScrapyDeprecationWarning)
|
||||
assert 'deprecated' in str(w[-1].message)
|
||||
hs.select("//div").extract()
|
||||
assert issubclass(w[1].category, ScrapyDeprecationWarning)
|
||||
assert 'deprecated' in str(w[-1].message)
|
||||
|
||||
def test_xpathselector(self):
|
||||
with warnings.catch_warnings(record=True):
|
||||
hs = XPathSelector(text=self.text)
|
||||
self.assertEqual(hs.select("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></div>'])
|
||||
self.assertRaises(RuntimeError, hs.css, 'div')
|
||||
|
||||
def test_htmlxpathselector(self):
|
||||
with warnings.catch_warnings(record=True):
|
||||
hs = HtmlXPathSelector(text=self.text)
|
||||
self.assertEqual(hs.select("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></div>'])
|
||||
self.assertRaises(RuntimeError, hs.css, 'div')
|
||||
|
||||
def test_xmlxpathselector(self):
|
||||
with warnings.catch_warnings(record=True):
|
||||
xs = XmlXPathSelector(text=self.text)
|
||||
self.assertEqual(xs.select("//div").extract(),
|
||||
[u'<div><img src="a.jpg"><p>Hello</p></img></div>'])
|
||||
self.assertRaises(RuntimeError, xs.css, 'div')
|
||||
|
|
|
|||
|
|
@ -114,22 +114,22 @@ class TranslatorMixinTest(unittest.TestCase):
|
|||
self.assertRaises(exc, self.c2x, css)
|
||||
|
||||
|
||||
class HTMLCSSSelectorTest(unittest.TestCase):
|
||||
class CSSSelectorTest(unittest.TestCase):
|
||||
|
||||
hcs_cls = Selector
|
||||
sscls = Selector
|
||||
|
||||
def setUp(self):
|
||||
self.htmlresponse = HtmlResponse('http://example.com', body=HTMLBODY)
|
||||
self.hcs = self.hcs_cls(self.htmlresponse)
|
||||
self.ss = self.sscls(self.htmlresponse)
|
||||
|
||||
def x(self, *a, **kw):
|
||||
return [v.strip() for v in self.hcs.css(*a, **kw).extract() if v.strip()]
|
||||
return [v.strip() for v in self.ss.css(*a, **kw).extract() if v.strip()]
|
||||
|
||||
def test_selector_simple(self):
|
||||
for x in self.hcs.css('input'):
|
||||
self.assertTrue(isinstance(x, self.hcs.__class__), x)
|
||||
self.assertEqual(self.hcs.css('input').extract(),
|
||||
[x.extract() for x in self.hcs.css('input')])
|
||||
for x in self.ss.css('input'):
|
||||
self.assertTrue(isinstance(x, self.ss.__class__), x)
|
||||
self.assertEqual(self.ss.css('input').extract(),
|
||||
[x.extract() for x in self.ss.css('input')])
|
||||
|
||||
def test_text_pseudo_element(self):
|
||||
self.assertEqual(self.x('#p-b2'), [u'<b id="p-b2">guy</b>'])
|
||||
|
|
@ -147,7 +147,7 @@ class HTMLCSSSelectorTest(unittest.TestCase):
|
|||
self.assertEqual(self.x('map[name="dummymap"] ::attr(shape)'), [u'circle', u'default'])
|
||||
|
||||
def test_nested_selector(self):
|
||||
self.assertEqual(self.hcs.css('p').css('b::text').extract(),
|
||||
self.assertEqual(self.ss.css('p').css('b::text').extract(),
|
||||
[u'hi', u'guy'])
|
||||
self.assertEqual(self.hcs.css('div').css('area:last-child').extract(),
|
||||
self.assertEqual(self.ss.css('div').css('area:last-child').extract(),
|
||||
[u'<area shape="default" id="area-nohref">'])
|
||||
|
|
@ -1,62 +0,0 @@
|
|||
"""
|
||||
Selectors tests, specific for lxml backend
|
||||
"""
|
||||
|
||||
import unittest
|
||||
from scrapy.tests import test_selector
|
||||
from scrapy.http import TextResponse, HtmlResponse, XmlResponse
|
||||
from scrapy.selector.lxmldocument import LxmlDocument
|
||||
from scrapy.selector.lxmlsel import XmlXPathSelector, HtmlXPathSelector, XPathSelector
|
||||
|
||||
|
||||
class LxmlXPathSelectorTestCase(test_selector.XPathSelectorTestCase):
|
||||
|
||||
xs_cls = XPathSelector
|
||||
hxs_cls = HtmlXPathSelector
|
||||
xxs_cls = XmlXPathSelector
|
||||
|
||||
def test_remove_namespaces(self):
|
||||
xml = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<feed xmlns="http://www.w3.org/2005/Atom" xml:lang="en-US" xmlns:media="http://search.yahoo.com/mrss/">
|
||||
<link type="text/html">
|
||||
<link type="application/atom+xml">
|
||||
</feed>
|
||||
"""
|
||||
xxs = XmlXPathSelector(XmlResponse("http://example.com/feed.atom", body=xml))
|
||||
self.assertEqual(len(xxs.select("//link")), 0)
|
||||
xxs.remove_namespaces()
|
||||
self.assertEqual(len(xxs.select("//link")), 2)
|
||||
|
||||
def test_remove_attributes_namespaces(self):
|
||||
xml = """<?xml version="1.0" encoding="UTF-8"?>
|
||||
<feed xmlns:atom="http://www.w3.org/2005/Atom" xml:lang="en-US" xmlns:media="http://search.yahoo.com/mrss/">
|
||||
<link atom:type="text/html">
|
||||
<link atom:type="application/atom+xml">
|
||||
</feed>
|
||||
"""
|
||||
xxs = XmlXPathSelector(XmlResponse("http://example.com/feed.atom", body=xml))
|
||||
self.assertEqual(len(xxs.select("//link/@type")), 0)
|
||||
xxs.remove_namespaces()
|
||||
self.assertEqual(len(xxs.select("//link/@type")), 2)
|
||||
|
||||
|
||||
class Libxml2DocumentTest(unittest.TestCase):
|
||||
|
||||
def test_caching(self):
|
||||
r1 = HtmlResponse('http://www.example.com', body='<html><head></head><body></body></html>')
|
||||
r2 = r1.copy()
|
||||
|
||||
doc1 = LxmlDocument(r1)
|
||||
doc2 = LxmlDocument(r1)
|
||||
doc3 = LxmlDocument(r2)
|
||||
|
||||
# make sure it's cached
|
||||
assert doc1 is doc2
|
||||
assert doc1 is not doc3
|
||||
|
||||
def test_null_char(self):
|
||||
# make sure bodies with null char ('\x00') don't raise a TypeError exception
|
||||
self.body_content = 'test problematic \x00 body'
|
||||
response = TextResponse('http://example.com/catalog/product/blabla-123',
|
||||
headers={'Content-Type': 'text/plain; charset=utf-8'}, body=self.body_content)
|
||||
LxmlDocument(response)
|
||||
|
|
@ -0,0 +1,26 @@
|
|||
import unittest
|
||||
from scrapy.selector.lxmldocument import LxmlDocument
|
||||
from scrapy.http import TextResponse, HtmlResponse
|
||||
|
||||
|
||||
class Libxml2DocumentTest(unittest.TestCase):
|
||||
|
||||
def test_caching(self):
|
||||
r1 = HtmlResponse('http://www.example.com', body='<html><head></head><body></body></html>')
|
||||
r2 = r1.copy()
|
||||
|
||||
doc1 = LxmlDocument(r1)
|
||||
doc2 = LxmlDocument(r1)
|
||||
doc3 = LxmlDocument(r2)
|
||||
|
||||
# make sure it's cached
|
||||
assert doc1 is doc2
|
||||
assert doc1 is not doc3
|
||||
|
||||
def test_null_char(self):
|
||||
# make sure bodies with null char ('\x00') don't raise a TypeError exception
|
||||
body = 'test problematic \x00 body'
|
||||
response = TextResponse('http://example.com/catalog/product/blabla-123',
|
||||
headers={'Content-Type': 'text/plain; charset=utf-8'},
|
||||
body=body)
|
||||
LxmlDocument(response)
|
||||
Loading…
Reference in New Issue