import weakref
import parsel
import pytest
from packaging import version
from twisted.trial import unittest
from scrapy.http import HtmlResponse, TextResponse, XmlResponse
from scrapy.selector import Selector
PARSEL_VERSION = version.parse(getattr(parsel, "__version__", "0.0"))
PARSEL_18_PLUS = PARSEL_VERSION >= version.parse("1.8.0")
class SelectorTestCase(unittest.TestCase):
def test_simple_selection(self):
"""Simple selector tests"""
body = b"
"
response = TextResponse(url="http://example.com", body=body, encoding="utf-8")
sel = Selector(response)
xl = sel.xpath("//input")
self.assertEqual(2, len(xl))
for x in xl:
assert isinstance(x, Selector)
self.assertEqual(
sel.xpath("//input").getall(), [x.get() for x in sel.xpath("//input")]
)
self.assertEqual(
[x.get() for x in sel.xpath("//input[@name='a']/@name")], ["a"]
)
self.assertEqual(
[
x.get()
for x in sel.xpath(
"number(concat(//input[@name='a']/@value, //input[@name='b']/@value))"
)
],
["12.0"],
)
self.assertEqual(sel.xpath("concat('xpath', 'rules')").getall(), ["xpathrules"])
self.assertEqual(
[
x.get()
for x in sel.xpath(
"concat(//input[@name='a']/@value, //input[@name='b']/@value)"
)
],
["12"],
)
def test_root_base_url(self):
body = b''
url = "http://example.com"
response = TextResponse(url=url, body=body, encoding="utf-8")
sel = Selector(response)
self.assertEqual(url, sel.root.base)
def test_flavor_detection(self):
text = b'
']
)
def test_http_header_encoding_precedence(self):
# '\xa3' = pound symbol in unicode
# '\xc2\xa3' = pound symbol in utf-8
# '\xa3' = pound symbol in latin-1 (iso-8859-1)
meta = (
''
)
head = f"{meta}"
body_content = '\xa3'
body = f"{body_content}"
html = f"{head}{body}"
encoding = "utf-8"
html_utf8 = html.encode(encoding)
headers = {"Content-Type": ["text/html; charset=utf-8"]}
response = HtmlResponse(
url="http://example.com", headers=headers, body=html_utf8
)
x = Selector(response)
self.assertEqual(x.xpath("//span[@id='blank']/text()").getall(), ["\xa3"])
def test_badly_encoded_body(self):
# \xe9 alone isn't valid utf8 sequence
r1 = TextResponse(
"http://www.example.com",
body=b"
an Jos\xe9 de
",
encoding="utf-8",
)
Selector(r1).xpath("//text()").getall()
def test_weakref_slots(self):
"""Check that classes are using slots and are weak-referenceable"""
x = Selector(text="")
weakref.ref(x)
assert not hasattr(x, "__dict__"), (
f"{x.__class__.__name__} does not use __slots__"
)
def test_selector_bad_args(self):
with self.assertRaisesRegex(ValueError, "received both response and text"):
Selector(TextResponse(url="http://example.com", body=b""), text="")
class JMESPathTestCase(unittest.TestCase):
@pytest.mark.skipif(
not PARSEL_18_PLUS, reason="parsel < 1.8 doesn't support jmespath"
)
def test_json_has_html(self) -> None:
"""Sometimes the information is returned in a json wrapper"""
body = """
{
"content": [
{
"name": "A",
"value": "a"
},
{
"name": {
"age": 18
},
"value": "b"
},
{
"name": "C",
"value": "c"
},
{
"name": "D",
"value": "