diff --git a/requirements.txt b/requirements.txt
index a05cd3680..368e9340b 100644
--- a/requirements.txt
+++ b/requirements.txt
@@ -7,3 +7,4 @@ queuelib
six>=1.5.2
PyDispatcher>=2.0.5
service_identity
+parsel>=0.9.5
diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py
index 0d37004fb..0b1d3b926 100644
--- a/scrapy/http/request/form.py
+++ b/scrapy/http/request/form.py
@@ -7,6 +7,7 @@ See documentation in docs/topics/request-response.rst
from six.moves.urllib.parse import urljoin, urlencode
import lxml.html
+from parsel.selector import create_root_node
import six
from scrapy.http.request import Request
from scrapy.utils.python import to_bytes, is_listlike
@@ -56,8 +57,8 @@ def _urlencode(seq, enc):
def _get_form(response, formname, formid, formnumber, formxpath):
"""Find the form element """
- from scrapy.selector.lxmldocument import LxmlDocument
- root = LxmlDocument(response, lxml.html.HTMLParser)
+ text = response.body_as_unicode()
+ root = create_root_node(text, lxml.html.HTMLParser, base_url=response.url)
forms = root.xpath('//form')
if not forms:
raise ValueError("No
'
- parser = parser_cls(recover=True, encoding='utf8')
- return etree.fromstring(body, parser=parser, base_url=url)
-
-
-class LxmlDocument(object_ref):
-
- cache = weakref.WeakKeyDictionary()
- __slots__ = ['__weakref__']
-
- def __new__(cls, response, parser=etree.HTMLParser):
- cache = cls.cache.setdefault(response, {})
- if parser not in cache:
- obj = object_ref.__new__(cls)
- cache[parser] = _factory(response, parser)
- return cache[parser]
-
- def __str__(self):
- return "" % self.root.tag
diff --git a/scrapy/selector/unified.py b/scrapy/selector/unified.py
index eed8f94f7..5d77f7624 100644
--- a/scrapy/selector/unified.py
+++ b/scrapy/selector/unified.py
@@ -2,43 +2,22 @@
XPath selectors based on lxml
"""
-from lxml import etree
-import six
-
-from scrapy.utils.misc import extract_regex
+import warnings
+from parsel import Selector as _ParselSelector
from scrapy.utils.trackref import object_ref
-from scrapy.utils.python import to_bytes, flatten, iflatten
-from scrapy.utils.decorators import deprecated
+from scrapy.utils.python import to_bytes
from scrapy.http import HtmlResponse, XmlResponse
-from .lxmldocument import LxmlDocument
-from .csstranslator import ScrapyHTMLTranslator, ScrapyGenericTranslator
+from scrapy.utils.decorators import deprecated
+from scrapy.exceptions import ScrapyDeprecationWarning
__all__ = ['Selector', 'SelectorList']
-class SafeXMLParser(etree.XMLParser):
- def __init__(self, *args, **kwargs):
- kwargs.setdefault('resolve_entities', False)
- super(SafeXMLParser, self).__init__(*args, **kwargs)
-
-_ctgroup = {
- 'html': {'_parser': etree.HTMLParser,
- '_csstranslator': ScrapyHTMLTranslator(),
- '_tostring_method': 'html'},
- 'xml': {'_parser': SafeXMLParser,
- '_csstranslator': ScrapyGenericTranslator(),
- '_tostring_method': 'xml'},
-}
-
-
def _st(response, st):
if st is None:
return 'xml' if isinstance(response, XmlResponse) else 'html'
- elif st in ('xml', 'html'):
- return st
- else:
- raise ValueError('Invalid type: %s' % st)
+ return st
def _response_from_text(text, st):
@@ -47,149 +26,7 @@ def _response_from_text(text, st):
body=to_bytes(text, 'utf-8'))
-class Selector(object_ref):
-
- __slots__ = ['response', 'text', 'namespaces', 'type', '_expr', '_root',
- '__weakref__', '_parser', '_csstranslator', '_tostring_method']
-
- _default_type = None
- _default_namespaces = {
- "re": "http://exslt.org/regular-expressions",
-
- # supported in libxslt:
- # set:difference
- # set:has-same-node
- # set:intersection
- # set:leading
- # set:trailing
- "set": "http://exslt.org/sets"
- }
- _lxml_smart_strings = False
-
- def __init__(self, response=None, text=None, type=None, namespaces=None,
- _root=None, _expr=None):
- self.type = st = _st(response, type or self._default_type)
- self._parser = _ctgroup[st]['_parser']
- self._csstranslator = _ctgroup[st]['_csstranslator']
- self._tostring_method = _ctgroup[st]['_tostring_method']
-
- if text is not None:
- response = _response_from_text(text, st)
-
- if response is not None:
- _root = LxmlDocument(response, self._parser)
-
- self.response = response
- self.namespaces = dict(self._default_namespaces)
- if namespaces is not None:
- self.namespaces.update(namespaces)
- self._root = _root
- self._expr = _expr
-
- def xpath(self, query):
- try:
- xpathev = self._root.xpath
- except AttributeError:
- return SelectorList([])
-
- try:
- result = xpathev(query, namespaces=self.namespaces,
- smart_strings=self._lxml_smart_strings)
- except etree.XPathError:
- msg = u"Invalid XPath: %s" % query
- raise ValueError(msg if six.PY3 else msg.encode("unicode_escape"))
-
- if type(result) is not list:
- result = [result]
-
- result = [self.__class__(_root=x, _expr=query,
- namespaces=self.namespaces,
- type=self.type)
- for x in result]
- return SelectorList(result)
-
- def css(self, query):
- return self.xpath(self._css2xpath(query))
-
- def _css2xpath(self, query):
- return self._csstranslator.css_to_xpath(query)
-
- def re(self, regex):
- return extract_regex(regex, self.extract())
-
- def extract(self):
- try:
- return etree.tostring(self._root,
- method=self._tostring_method,
- encoding="unicode",
- with_tail=False)
- except (AttributeError, TypeError):
- if self._root is True:
- return u'1'
- elif self._root is False:
- return u'0'
- else:
- return six.text_type(self._root)
-
- def register_namespace(self, prefix, uri):
- if self.namespaces is None:
- self.namespaces = {}
- self.namespaces[prefix] = uri
-
- def remove_namespaces(self):
- for el in self._root.iter('*'):
- if el.tag.startswith('{'):
- el.tag = el.tag.split('}', 1)[1]
- # loop on element attributes also
- for an in el.attrib.keys():
- if an.startswith('{'):
- el.attrib[an.split('}', 1)[1]] = el.attrib.pop(an)
-
- def __nonzero__(self):
- return bool(self.extract())
-
- def __str__(self):
- data = repr(self.extract()[:40])
- return "<%s xpath=%r data=%s>" % (type(self).__name__, self._expr, data)
- __repr__ = __str__
-
- # Deprecated api
- @deprecated(use_instead='.xpath()')
- def select(self, xpath):
- return self.xpath(xpath)
-
- @deprecated(use_instead='.extract()')
- def extract_unquoted(self):
- return self.extract()
-
-
-class SelectorList(list):
-
- def __getslice__(self, i, j):
- return self.__class__(list.__getslice__(self, i, j))
-
- def xpath(self, xpath):
- return self.__class__(flatten([x.xpath(xpath) for x in self]))
-
- def css(self, xpath):
- return self.__class__(flatten([x.css(xpath) for x in self]))
-
- def re(self, regex):
- return flatten([x.re(regex) for x in self])
-
- def re_first(self, regex):
- for el in iflatten(x.re(regex) for x in self):
- return el
-
- def extract(self):
- return [x.extract() for x in self]
-
- def extract_first(self, default=None):
- for x in self:
- return x.extract()
- else:
- return default
-
+class SelectorList(_ParselSelector.selectorlist_cls, object_ref):
@deprecated(use_instead='.extract()')
def extract_unquoted(self):
return [x.extract_unquoted() for x in self]
@@ -201,3 +38,45 @@ class SelectorList(list):
@deprecated(use_instead='.xpath()')
def select(self, xpath):
return self.xpath(xpath)
+
+
+class Selector(_ParselSelector, object_ref):
+
+ __slots__ = ['response']
+ selectorlist_cls = SelectorList
+
+ def __init__(self, response=None, text=None, type=None, root=None, _root=None, **kwargs):
+ st = _st(response, type or self._default_type)
+
+ if _root is not None:
+ warnings.warn("Argument `_root` is deprecated, use `root` instead",
+ ScrapyDeprecationWarning, stacklevel=2)
+ if root is None:
+ root = _root
+ else:
+ warnings.warn("Ignoring deprecated `_root` argument, using provided `root`")
+
+ if text is not None:
+ response = _response_from_text(text, st)
+
+ if response is not None:
+ text = response.body_as_unicode()
+ kwargs.setdefault('base_url', response.url)
+
+ self.response = response
+ super(Selector, self).__init__(text=text, type=st, root=root, **kwargs)
+
+ # Deprecated api
+ @property
+ def _root(self):
+ warnings.warn("Attribute `_root` is deprecated, use `root` instead",
+ ScrapyDeprecationWarning, stacklevel=2)
+ return self.root
+
+ @deprecated(use_instead='.xpath()')
+ def select(self, xpath):
+ return self.xpath(xpath)
+
+ @deprecated(use_instead='.extract()')
+ def extract_unquoted(self):
+ return self.extract()
diff --git a/setup.py b/setup.py
index 7214f0dc6..4eb8d2318 100644
--- a/setup.py
+++ b/setup.py
@@ -44,6 +44,7 @@ setup(
'pyOpenSSL',
'cssselect>=0.9',
'six>=1.5.2',
+ 'parsel>=0.9.3',
'PyDispatcher>=2.0.5',
'service_identity',
],
diff --git a/tests/py3-ignores.txt b/tests/py3-ignores.txt
index 9be3a99a8..5a009db36 100644
--- a/tests/py3-ignores.txt
+++ b/tests/py3-ignores.txt
@@ -25,9 +25,6 @@ tests/test_mail.py
tests/test_pipeline_files.py
tests/test_pipeline_images.py
tests/test_proxy_connect.py
-tests/test_selector_csstranslator.py
-tests/test_selector_lxmldocument.py
-tests/test_selector.py
tests/test_spidermiddleware_depth.py
tests/test_spidermiddleware_httperror.py
tests/test_spidermiddleware_offsite.py
diff --git a/tests/test_selector.py b/tests/test_selector.py
index ad6a0e21c..141455b66 100644
--- a/tests/test_selector.py
+++ b/tests/test_selector.py
@@ -1,28 +1,24 @@
-import re
import warnings
import weakref
-import six
from twisted.trial import unittest
-from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.http import TextResponse, HtmlResponse, XmlResponse
from scrapy.selector import Selector
from scrapy.selector.lxmlsel import XmlXPathSelector, HtmlXPathSelector, XPathSelector
+from lxml import etree
class SelectorTestCase(unittest.TestCase):
- sscls = Selector
-
def test_simple_selection(self):
"""Simple selector tests"""
- body = ""
- response = TextResponse(url="http://example.com", body=body)
- sel = self.sscls(response)
+ body = b""
+ response = TextResponse(url="http://example.com", body=body, encoding='utf-8')
+ sel = Selector(response)
xl = sel.xpath('//input')
self.assertEqual(2, len(xl))
for x in xl:
- assert isinstance(x, self.sscls)
+ assert isinstance(x, Selector)
self.assertEqual(sel.xpath('//input').extract(),
[x.extract() for x in sel.xpath('//input')])
@@ -37,234 +33,41 @@ class SelectorTestCase(unittest.TestCase):
self.assertEqual([x.extract() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
[u'12'])
- def test_representation_slice(self):
- body = u"".format(50 * 'b')
- response = TextResponse(url="http://example.com", body=body, encoding='utf8')
- sel = self.sscls(response)
+ def test_root_base_url(self):
+ body = b''
+ url = "http://example.com"
+ response = TextResponse(url=url, body=body, encoding='utf-8')
+ sel = Selector(response)
+ self.assertEqual(url, sel.root.base)
- self.assertEqual(
- map(repr, sel.xpath('//input/@name')),
- ["".format(40 * 'b')]
- )
+ def test_deprecated_root_argument(self):
+ with warnings.catch_warnings(record=True) as w:
+ root = etree.fromstring(u'')
+ sel = Selector(_root=root)
+ self.assertIs(root, sel.root)
+ self.assertEqual(str(w[-1].message),
+ 'Argument `_root` is deprecated, use `root` instead')
- def test_representation_unicode_query(self):
- body = u"".format(50 * 'b')
- response = TextResponse(url="http://example.com", body=body, encoding='utf8')
- sel = self.sscls(response)
- self.assertEqual(
- map(repr, sel.xpath(u'//input[@value="\xa9"]/@value')),
- [""]
- )
-
- def test_extract_first(self):
- """Test if extract_first() returns first element"""
- body = ''
- response = TextResponse(url="http://example.com", body=body)
- sel = self.sscls(response)
-
- self.assertEqual(sel.xpath('//ul/li/text()').extract_first(),
- sel.xpath('//ul/li/text()').extract()[0])
-
- self.assertEqual(sel.xpath('//ul/li[@id="1"]/text()').extract_first(),
- sel.xpath('//ul/li[@id="1"]/text()').extract()[0])
-
- self.assertEqual(sel.xpath('//ul/li[2]/text()').extract_first(),
- sel.xpath('//ul/li/text()').extract()[1])
-
- self.assertEqual(sel.xpath('/ul/li[@id="doesnt-exist"]/text()').extract_first(), None)
-
- def test_extract_first_default(self):
- """Test if extract_first() returns default value when no results found"""
- body = ''
- response = TextResponse(url="http://example.com", body=body)
- sel = self.sscls(response)
-
- self.assertEqual(sel.xpath('//div/text()').extract_first(default='missing'), 'missing')
-
- def test_re_first(self):
- """Test if re_first() returns first matched element"""
- body = ''
- response = TextResponse(url="http://example.com", body=body)
- sel = self.sscls(response)
-
- self.assertEqual(sel.xpath('//ul/li/text()').re_first('\d'),
- sel.xpath('//ul/li/text()').re('\d')[0])
-
- self.assertEqual(sel.xpath('//ul/li[@id="1"]/text()').re_first('\d'),
- sel.xpath('//ul/li[@id="1"]/text()').re('\d')[0])
-
- self.assertEqual(sel.xpath('//ul/li[2]/text()').re_first('\d'),
- sel.xpath('//ul/li/text()').re('\d')[1])
-
- self.assertEqual(sel.xpath('/ul/li/text()').re_first('\w+'), None)
- self.assertEqual(sel.xpath('/ul/li[@id="doesnt-exist"]/text()').re_first('\d'), None)
-
- def test_select_unicode_query(self):
- body = u""
- response = TextResponse(url="http://example.com", body=body, encoding='utf8')
- sel = self.sscls(response)
- self.assertEqual(sel.xpath(u'//input[@name="\xa9"]/@value').extract(), [u'1'])
-
- def test_list_elements_type(self):
- """Test Selector returning the same type in selection methods"""
- text = 'test
'
- assert isinstance(self.sscls(text=text).xpath("//p")[0], self.sscls)
- assert isinstance(self.sscls(text=text).css("p")[0], self.sscls)
-
- def test_boolean_result(self):
- body = "
"
- response = TextResponse(url="http://example.com", body=body)
- xs = self.sscls(response)
- self.assertEquals(xs.xpath("//input[@name='a']/@name='a'").extract(), [u'1'])
- self.assertEquals(xs.xpath("//input[@name='a']/@name='n'").extract(), [u'0'])
-
- def test_differences_parsing_xml_vs_html(self):
- """Test that XML and HTML Selector's behave differently"""
- # some text which is parsed differently by XML and HTML flavors
- text = '
Hello
'
- hs = self.sscls(text=text, type='html')
- self.assertEqual(hs.xpath("//div").extract(),
- [u'
Hello
'])
-
- xs = self.sscls(text=text, type='xml')
- self.assertEqual(xs.xpath("//div").extract(),
- [u'
Hello
'])
+ def test_deprecated_root_argument_ambiguous(self):
+ with warnings.catch_warnings(record=True) as w:
+ _root = etree.fromstring(u'')
+ root = etree.fromstring(u'')
+ sel = Selector(_root=_root, root=root)
+ self.assertIs(root, sel.root)
+ self.assertIn('Ignoring deprecated `_root` argument', str(w[-1].message))
def test_flavor_detection(self):
- text = '
Hello
'
- sel = self.sscls(XmlResponse('http://example.com', body=text))
+ text = b'
Hello
'
+ sel = Selector(XmlResponse('http://example.com', body=text, encoding='utf-8'))
self.assertEqual(sel.type, 'xml')
self.assertEqual(sel.xpath("//div").extract(),
[u'
Hello
'])
- sel = self.sscls(HtmlResponse('http://example.com', body=text))
+ sel = Selector(HtmlResponse('http://example.com', body=text, encoding='utf-8'))
self.assertEqual(sel.type, 'html')
self.assertEqual(sel.xpath("//div").extract(),
[u'
Hello
'])
- def test_nested_selectors(self):
- """Nested selector tests"""
- body = """
-
-
- """
-
- response = HtmlResponse(url="http://example.com", body=body)
- x = self.sscls(response)
- divtwo = x.xpath('//div[@class="two"]')
- self.assertEqual(divtwo.xpath("//li").extract(),
- ["one", "two", "four", "five", "six"])
- self.assertEqual(divtwo.xpath("./ul/li").extract(),
- ["four", "five", "six"])
- self.assertEqual(divtwo.xpath(".//li").extract(),
- ["four", "five", "six"])
- self.assertEqual(divtwo.xpath("./li").extract(), [])
-
- def test_mixed_nested_selectors(self):
- body = '''
- notme
-
- '''
- sel = self.sscls(text=body)
- self.assertEqual(sel.xpath('//div[@id="1"]').css('span::text').extract(), [u'me'])
- self.assertEqual(sel.css('#1').xpath('./span/text()').extract(), [u'me'])
-
- def test_dont_strip(self):
- sel = self.sscls(text='')
- self.assertEqual(sel.xpath("//text()").extract(), [u'fff: ', u'zzz'])
-
- def test_namespaces_simple(self):
- body = """
-
- take this
- found
-
- """
-
- response = XmlResponse(url="http://example.com", body=body)
- x = self.sscls(response)
-
- x.register_namespace("somens", "http://scrapy.org")
- self.assertEqual(x.xpath("//somens:a/text()").extract(),
- [u'take this'])
-
- def test_namespaces_multiple(self):
- body = """
-
- hello
- value
- iron90Dried Rose
-
- """
- response = XmlResponse(url="http://example.com", body=body)
- x = self.sscls(response)
- x.register_namespace("xmlns", "http://webservices.amazon.com/AWSECommerceService/2005-10-05")
- x.register_namespace("p", "http://www.scrapy.org/product")
- x.register_namespace("b", "http://somens.com")
- self.assertEqual(len(x.xpath("//xmlns:TestTag")), 1)
- self.assertEqual(x.xpath("//b:Operation/text()").extract()[0], 'hello')
- self.assertEqual(x.xpath("//xmlns:TestTag/@b:att").extract()[0], 'value')
- self.assertEqual(x.xpath("//p:SecondTestTag/xmlns:price/text()").extract()[0], '90')
- self.assertEqual(x.xpath("//p:SecondTestTag").xpath("./xmlns:price/text()")[0].extract(), '90')
- self.assertEqual(x.xpath("//p:SecondTestTag/xmlns:material/text()").extract()[0], 'iron')
-
- def test_re(self):
- body = """Name: Mary
-
- - Name: John
- - Age: 10
- - Name: Paul
- - Age: 20
-
- Age: 20
-
"""
- response = HtmlResponse(url="http://example.com", body=body)
- x = self.sscls(response)
-
- name_re = re.compile("Name: (\w+)")
- self.assertEqual(x.xpath("//ul/li").re(name_re),
- ["John", "Paul"])
- self.assertEqual(x.xpath("//ul/li").re("Age: (\d+)"),
- ["10", "20"])
-
- def test_re_intl(self):
- body = """Evento: cumplea\xc3\xb1os
"""
- response = HtmlResponse(url="http://example.com", body=body, encoding='utf-8')
- x = self.sscls(response)
- self.assertEqual(x.xpath("//div").re("Evento: (\w+)"), [u'cumplea\xf1os'])
-
- def test_selector_over_text(self):
- hs = self.sscls(text='lala')
- self.assertEqual(hs.extract(), u'lala')
- xs = self.sscls(text='lala', type='xml')
- self.assertEqual(xs.extract(), u'lala')
- self.assertEqual(xs.xpath('.').extract(), [u'lala'])
-
- def test_invalid_xpath(self):
- "Test invalid xpath raises ValueError with the invalid xpath"
- response = XmlResponse(url="http://example.com", body="")
- x = self.sscls(response)
- xpath = "//test[@foo='bar]"
- self.assertRaisesRegexp(ValueError, re.escape(xpath), x.xpath, xpath)
-
- def test_invalid_xpath_unicode(self):
- "Test *Unicode* invalid xpath raises ValueError with the invalid xpath"
- response = XmlResponse(url="http://example.com", body="")
- x = self.sscls(response)
- xpath = u"//test[@foo='\u0431ar]"
- encoded = xpath if six.PY3 else xpath.encode('unicode_escape')
- self.assertRaisesRegexp(ValueError, re.escape(encoded), x.xpath, xpath)
-
def test_http_header_encoding_precedence(self):
# u'\xa3' = pound symbol in unicode
# u'\xc2\xa3' = pound symbol in utf-8
@@ -280,131 +83,45 @@ class SelectorTestCase(unittest.TestCase):
headers = {'Content-Type': ['text/html; charset=utf-8']}
response = HtmlResponse(url="http://example.com", headers=headers, body=html_utf8)
- x = self.sscls(response)
+ x = Selector(response)
self.assertEquals(x.xpath("//span[@id='blank']/text()").extract(),
[u'\xa3'])
- def test_empty_bodies(self):
- # shouldn't raise errors
- r1 = TextResponse('http://www.example.com', body='')
- self.sscls(r1).xpath('//text()').extract()
-
- def test_null_bytes(self):
- # shouldn't raise errors
- r1 = TextResponse('http://www.example.com', \
- body='pre\x00post', \
- encoding='utf-8')
- self.sscls(r1).xpath('//text()').extract()
-
def test_badly_encoded_body(self):
# \xe9 alone isn't valid utf8 sequence
r1 = TextResponse('http://www.example.com', \
- body='an Jos\xe9 de
', \
+ body=b'an Jos\xe9 de
', \
encoding='utf-8')
- self.sscls(r1).xpath('//text()').extract()
-
- def test_select_on_unevaluable_nodes(self):
- r = self.sscls(text=u'some text')
- # Text node
- x1 = r.xpath('//text()')
- self.assertEquals(x1.extract(), [u'some text'])
- self.assertEquals(x1.xpath('.//b').extract(), [])
- # Tag attribute
- x1 = r.xpath('//span/@class')
- self.assertEquals(x1.extract(), [u'big'])
- self.assertEquals(x1.xpath('.//text()').extract(), [])
-
- def test_select_on_text_nodes(self):
- r = self.sscls(text=u'Options:opt1
Otheropt2
')
- x1 = r.xpath("//div/descendant::text()[preceding-sibling::b[contains(text(), 'Options')]]")
- self.assertEquals(x1.extract(), [u'opt1'])
-
- x1 = r.xpath("//div/descendant::text()/preceding-sibling::b[contains(text(), 'Options')]")
- self.assertEquals(x1.extract(), [u'Options:'])
-
- def test_nested_select_on_text_nodes(self):
- # FIXME: does not work with lxml backend [upstream]
- r = self.sscls(text=u'Options:opt1
Otheropt2
')
- x1 = r.xpath("//div/descendant::text()")
- x2 = x1.xpath("./preceding-sibling::b[contains(text(), 'Options')]")
- self.assertEquals(x2.extract(), [u'Options:'])
- test_nested_select_on_text_nodes.skip = "Text nodes lost parent node reference in lxml"
+ Selector(r1).xpath('//text()').extract()
def test_weakref_slots(self):
"""Check that classes are using slots and are weak-referenceable"""
- x = self.sscls()
+ x = Selector(text='')
weakref.ref(x)
assert not hasattr(x, '__dict__'), "%s does not use __slots__" % \
x.__class__.__name__
- def test_remove_namespaces(self):
- xml = """
-
-
-
-
-"""
- sel = self.sscls(XmlResponse("http://example.com/feed.atom", body=xml))
- self.assertEqual(len(sel.xpath("//link")), 0)
- sel.remove_namespaces()
- self.assertEqual(len(sel.xpath("//link")), 2)
+ def test_deprecated_selector_methods(self):
+ sel = Selector(TextResponse(url="http://example.com", body=b'some text
'))
- def test_remove_attributes_namespaces(self):
- xml = """
-
-
-
-
-"""
- sel = self.sscls(XmlResponse("http://example.com/feed.atom", body=xml))
- self.assertEqual(len(sel.xpath("//link/@type")), 0)
- sel.remove_namespaces()
- self.assertEqual(len(sel.xpath("//link/@type")), 2)
+ with warnings.catch_warnings(record=True) as w:
+ sel.select('//p')
+ self.assertSubstring('Use .xpath() instead', str(w[-1].message))
- def test_smart_strings(self):
- """Lxml smart strings return values"""
+ with warnings.catch_warnings(record=True) as w:
+ sel.extract_unquoted()
+ self.assertSubstring('Use .extract() instead', str(w[-1].message))
- class SmartStringsSelector(Selector):
- _lxml_smart_strings = True
+ def test_deprecated_selectorlist_methods(self):
+ sel = Selector(TextResponse(url="http://example.com", body=b'some text
'))
- body = """
-
-
- """
+ with warnings.catch_warnings(record=True) as w:
+ sel.xpath('//p').select('.')
+ self.assertSubstring('Use .xpath() instead', str(w[-1].message))
- response = HtmlResponse(url="http://example.com", body=body)
-
- # .getparent() is available for text nodes and attributes
- # only when smart_strings are on
- x = self.sscls(response)
- li_text = x.xpath('//li/text()')
- self.assertFalse(any(map(lambda e: hasattr(e._root, 'getparent'), li_text)))
- div_class = x.xpath('//div/@class')
- self.assertFalse(any(map(lambda e: hasattr(e._root, 'getparent'), div_class)))
-
- x = SmartStringsSelector(response)
- li_text = x.xpath('//li/text()')
- self.assertTrue(all(map(lambda e: hasattr(e._root, 'getparent'), li_text)))
- div_class = x.xpath('//div/@class')
- self.assertTrue(all(map(lambda e: hasattr(e._root, 'getparent'), div_class)))
-
- def test_xml_entity_expansion(self):
- malicious_xml = ''\
- ' ]>&xxe;'
-
- response = XmlResponse('http://example.com', body=malicious_xml)
- sel = self.sscls(response=response)
-
- self.assertEqual(sel.extract(), '&xxe;')
+ with warnings.catch_warnings(record=True) as w:
+ sel.xpath('//p').extract_unquoted()
+ self.assertSubstring('Use .extract() instead', str(w[-1].message))
class DeprecatedXpathSelectorTest(unittest.TestCase):
@@ -488,146 +205,3 @@ class DeprecatedXpathSelectorTest(unittest.TestCase):
self.assertTrue(isinstance(usel, Selector))
self.assertTrue(isinstance(sel, XPathSelector))
self.assertTrue(isinstance(usel, XPathSelector))
-
- def test_xpathselector(self):
- with warnings.catch_warnings():
- warnings.simplefilter('ignore', ScrapyDeprecationWarning)
- hs = XPathSelector(text=self.text)
- self.assertEqual(hs.select("//div").extract(),
- [u'
Hello
'])
- self.assertRaises(RuntimeError, hs.css, 'div')
-
- def test_htmlxpathselector(self):
- with warnings.catch_warnings():
- warnings.simplefilter('ignore', ScrapyDeprecationWarning)
- hs = HtmlXPathSelector(text=self.text)
- self.assertEqual(hs.select("//div").extract(),
- [u'
Hello
'])
- self.assertRaises(RuntimeError, hs.css, 'div')
-
- def test_xmlxpathselector(self):
- with warnings.catch_warnings():
- warnings.simplefilter('ignore', ScrapyDeprecationWarning)
- xs = XmlXPathSelector(text=self.text)
- self.assertEqual(xs.select("//div").extract(),
- [u'
Hello
'])
- self.assertRaises(RuntimeError, xs.css, 'div')
-
-
-class ExsltTestCase(unittest.TestCase):
-
- sscls = Selector
-
- def test_regexp(self):
- """EXSLT regular expression tests"""
- body = """
-
-
- """
- response = TextResponse(url="http://example.com", body=body)
- sel = self.sscls(response)
-
- # re:test()
- self.assertEqual(
- sel.xpath(
- '//input[re:test(@name, "[A-Z]+", "i")]').extract(),
- [x.extract() for x in sel.xpath('//input[re:test(@name, "[A-Z]+", "i")]')])
- self.assertEqual(
- [x.extract()
- for x in sel.xpath(
- '//a[re:test(@href, "\.html$")]/text()')],
- [u'first link', u'second link'])
- self.assertEqual(
- [x.extract()
- for x in sel.xpath(
- '//a[re:test(@href, "first")]/text()')],
- [u'first link'])
- self.assertEqual(
- [x.extract()
- for x in sel.xpath(
- '//a[re:test(@href, "second")]/text()')],
- [u'second link'])
-
-
- # re:match() is rather special: it returns a node-set of nodes
- #[u'http://www.bayes.co.uk/xml/index.xml?/xml/utils/rechecker.xml',
- #u'http',
- #u'www.bayes.co.uk',
- #u'',
- #u'/xml/index.xml?/xml/utils/rechecker.xml']
- self.assertEqual(
- sel.xpath('re:match(//a[re:test(@href, "\.xml$")]/@href,'
- '"(\w+):\/\/([^/:]+)(:\d*)?([^# ]*)")/text()').extract(),
- [u'http://www.bayes.co.uk/xml/index.xml?/xml/utils/rechecker.xml',
- u'http',
- u'www.bayes.co.uk',
- u'',
- u'/xml/index.xml?/xml/utils/rechecker.xml'])
-
-
-
- # re:replace()
- self.assertEqual(
- sel.xpath('re:replace(//a[re:test(@href, "\.xml$")]/@href,'
- '"(\w+)://(.+)(\.xml)", "","https://\\2.html")').extract(),
- [u'https://www.bayes.co.uk/xml/index.xml?/xml/utils/rechecker.html'])
-
- def test_set(self):
- """EXSLT set manipulation tests"""
- # microdata example from http://schema.org/Event
- body="""
-
- """
- response = TextResponse(url="http://example.com", body=body)
- sel = self.sscls(response)
-
- self.assertEqual(
- sel.xpath('''//div[@itemtype="http://schema.org/Event"]
- //@itemprop''').extract(),
- [u'url',
- u'name',
- u'startDate',
- u'location',
- u'url',
- u'address',
- u'addressLocality',
- u'addressRegion',
- u'offers',
- u'lowPrice',
- u'offerCount']
- )
-
- self.assertEqual(sel.xpath('''
- set:difference(//div[@itemtype="http://schema.org/Event"]
- //@itemprop,
- //div[@itemtype="http://schema.org/Event"]
- //*[@itemscope]/*/@itemprop)''').extract(),
- [u'url', u'name', u'startDate', u'location', u'offers'])
diff --git a/tests/test_selector_csstranslator.py b/tests/test_selector_csstranslator.py
index 1bc8882f8..2d82fcba7 100644
--- a/tests/test_selector_csstranslator.py
+++ b/tests/test_selector_csstranslator.py
@@ -1,153 +1,22 @@
"""
Selector tests for cssselect backend
"""
+import warnings
from twisted.trial import unittest
-from scrapy.http import HtmlResponse
-from scrapy.selector.csstranslator import ScrapyHTMLTranslator
-from scrapy.selector import Selector
-from cssselect.parser import SelectorSyntaxError
-from cssselect.xpath import ExpressionError
+from scrapy.selector.csstranslator import (
+ ScrapyHTMLTranslator,
+ ScrapyGenericTranslator,
+ ScrapyXPathExpr
+)
-HTMLBODY = b'''
-
-
-
-
-
-'''
+class DeprecatedClassesTest(unittest.TestCase):
+
+ def test_deprecated_warnings(self):
+ for cls in [ScrapyHTMLTranslator, ScrapyGenericTranslator, ScrapyXPathExpr]:
+ with warnings.catch_warnings(record=True) as w:
+ obj = cls()
+ self.assertIn('%s is deprecated' % cls.__name__, str(w[-1].message),
+ 'Missing deprecate warning for %s' % cls.__name__)
-class TranslatorMixinTest(unittest.TestCase):
-
- tr_cls = ScrapyHTMLTranslator
-
- def setUp(self):
- self.tr = self.tr_cls()
- self.c2x = self.tr.css_to_xpath
-
- def test_attr_function(self):
- cases = [
- ('::attr(name)', u'descendant-or-self::*/@name'),
- ('a::attr(href)', u'descendant-or-self::a/@href'),
- ('a ::attr(img)', u'descendant-or-self::a/descendant-or-self::*/@img'),
- ('a > ::attr(class)', u'descendant-or-self::a/*/@class'),
- ]
- for css, xpath in cases:
- self.assertEqual(self.c2x(css), xpath, css)
-
- def test_attr_function_exception(self):
- cases = [
- ('::attr(12)', ExpressionError),
- ('::attr(34test)', ExpressionError),
- ('::attr(@href)', SelectorSyntaxError),
- ]
- for css, exc in cases:
- self.assertRaises(exc, self.c2x, css)
-
- def test_text_pseudo_element(self):
- cases = [
- ('::text', u'descendant-or-self::text()'),
- ('p::text', u'descendant-or-self::p/text()'),
- ('p ::text', u'descendant-or-self::p/descendant-or-self::text()'),
- ('#id::text', u"descendant-or-self::*[@id = 'id']/text()"),
- ('p#id::text', u"descendant-or-self::p[@id = 'id']/text()"),
- ('p#id ::text', u"descendant-or-self::p[@id = 'id']/descendant-or-self::text()"),
- ('p#id > ::text', u"descendant-or-self::p[@id = 'id']/*/text()"),
- ('p#id ~ ::text', u"descendant-or-self::p[@id = 'id']/following-sibling::*/text()"),
- ('a[href]::text', u'descendant-or-self::a[@href]/text()'),
- ('a[href] ::text', u'descendant-or-self::a[@href]/descendant-or-self::text()'),
- ('p::text, a::text', u"descendant-or-self::p/text() | descendant-or-self::a/text()"),
- ]
- for css, xpath in cases:
- self.assertEqual(self.c2x(css), xpath, css)
-
- def test_pseudo_function_exception(self):
- cases = [
- ('::attribute(12)', ExpressionError),
- ('::text()', ExpressionError),
- ('::attr(@href)', SelectorSyntaxError),
- ]
- for css, exc in cases:
- self.assertRaises(exc, self.c2x, css)
-
- def test_unknown_pseudo_element(self):
- cases = [
- ('::text-node', ExpressionError),
- ]
- for css, exc in cases:
- self.assertRaises(exc, self.c2x, css)
-
- def test_unknown_pseudo_class(self):
- cases = [
- (':text', ExpressionError),
- (':attribute(name)', ExpressionError),
- ]
- for css, exc in cases:
- self.assertRaises(exc, self.c2x, css)
-
-
-class CSSSelectorTest(unittest.TestCase):
-
- sscls = Selector
-
- def setUp(self):
- self.htmlresponse = HtmlResponse('http://example.com', body=HTMLBODY)
- self.sel = self.sscls(self.htmlresponse)
-
- def x(self, *a, **kw):
- return [v.strip() for v in self.sel.css(*a, **kw).extract() if v.strip()]
-
- def test_selector_simple(self):
- for x in self.sel.css('input'):
- self.assertTrue(isinstance(x, self.sel.__class__), x)
- self.assertEqual(self.sel.css('input').extract(),
- [x.extract() for x in self.sel.css('input')])
-
- def test_text_pseudo_element(self):
- self.assertEqual(self.x('#p-b2'), [u'guy'])
- self.assertEqual(self.x('#p-b2::text'), [u'guy'])
- self.assertEqual(self.x('#p-b2 ::text'), [u'guy'])
- self.assertEqual(self.x('#paragraph::text'), [u'lorem ipsum text'])
- self.assertEqual(self.x('#paragraph ::text'), [u'lorem ipsum text', u'hi', u'there', u'guy'])
- self.assertEqual(self.x('p::text'), [u'lorem ipsum text'])
- self.assertEqual(self.x('p ::text'), [u'lorem ipsum text', u'hi', u'there', u'guy'])
-
- def test_attribute_function(self):
- self.assertEqual(self.x('#p-b2::attr(id)'), [u'p-b2'])
- self.assertEqual(self.x('.cool-footer::attr(class)'), [u'cool-footer'])
- self.assertEqual(self.x('.cool-footer ::attr(id)'), [u'foobar-div', u'foobar-span'])
- self.assertEqual(self.x('map[name="dummymap"] ::attr(shape)'), [u'circle', u'default'])
-
- def test_nested_selector(self):
- self.assertEqual(self.sel.css('p').css('b::text').extract(),
- [u'hi', u'guy'])
- self.assertEqual(self.sel.css('div').css('area:last-child').extract(),
- [u''])
diff --git a/tests/test_selector_lxmldocument.py b/tests/test_selector_lxmldocument.py
deleted file mode 100644
index 090cc21bc..000000000
--- a/tests/test_selector_lxmldocument.py
+++ /dev/null
@@ -1,26 +0,0 @@
-import unittest
-from scrapy.selector.lxmldocument import LxmlDocument
-from scrapy.http import TextResponse, HtmlResponse
-
-
-class LxmlDocumentTest(unittest.TestCase):
-
- def test_caching(self):
- r1 = HtmlResponse('http://www.example.com', body=b'')
- r2 = r1.copy()
-
- doc1 = LxmlDocument(r1)
- doc2 = LxmlDocument(r1)
- doc3 = LxmlDocument(r2)
-
- # make sure it's cached
- assert doc1 is doc2
- assert doc1 is not doc3
-
- def test_null_char(self):
- # make sure bodies with null char ('\x00') don't raise a TypeError exception
- body = b'test problematic \x00 body'
- response = TextResponse('http://example.com/catalog/product/blabla-123',
- headers={'Content-Type': 'text/plain; charset=utf-8'},
- body=body)
- LxmlDocument(response)