import weakref import parsel import pytest from packaging import version from scrapy.http import HtmlResponse, TextResponse, XmlResponse from scrapy.selector import Selector PARSEL_VERSION = version.parse(getattr(parsel, "__version__", "0.0")) PARSEL_18_PLUS = PARSEL_VERSION >= version.parse("1.8.0") class TestSelector: def test_simple_selection(self): """Simple selector tests""" body = b"

" response = TextResponse(url="http://example.com", body=body, encoding="utf-8") sel = Selector(response) xl = sel.xpath("//input") assert len(xl) == 2 for x in xl: assert isinstance(x, Selector) assert sel.xpath("//input").getall() == [x.get() for x in sel.xpath("//input")] assert [x.get() for x in sel.xpath("//input[@name='a']/@name")] == ["a"] assert [ x.get() for x in sel.xpath( "number(concat(//input[@name='a']/@value, //input[@name='b']/@value))" ) ] == ["12.0"] assert sel.xpath("concat('xpath', 'rules')").getall() == ["xpathrules"] assert [ x.get() for x in sel.xpath( "concat(//input[@name='a']/@value, //input[@name='b']/@value)" ) ] == ["12"] def test_root_base_url(self): body = b'
' url = "http://example.com" response = TextResponse(url=url, body=body, encoding="utf-8") sel = Selector(response) assert url == sel.root.base def test_flavor_detection(self): text = b'

Hello

' sel = Selector(XmlResponse("http://example.com", body=text, encoding="utf-8")) assert sel.type == "xml" assert sel.xpath("//div").getall() == [ '

Hello

' ] sel = Selector(HtmlResponse("http://example.com", body=text, encoding="utf-8")) assert sel.type == "html" assert sel.xpath("//div").getall() == [ '

Hello

' ] def test_http_header_encoding_precedence(self): # '\xa3' = pound symbol in unicode # '\xc2\xa3' = pound symbol in utf-8 # '\xa3' = pound symbol in latin-1 (iso-8859-1) meta = ( '' ) head = f"{meta}" body_content = '\xa3' body = f"{body_content}" html = f"{head}{body}" encoding = "utf-8" html_utf8 = html.encode(encoding) headers = {"Content-Type": ["text/html; charset=utf-8"]} response = HtmlResponse( url="http://example.com", headers=headers, body=html_utf8 ) x = Selector(response) assert x.xpath("//span[@id='blank']/text()").getall() == ["\xa3"] def test_badly_encoded_body(self): # \xe9 alone isn't valid utf8 sequence r1 = TextResponse( "http://www.example.com", body=b"

an Jos\xe9 de

", encoding="utf-8", ) Selector(r1).xpath("//text()").getall() def test_weakref_slots(self): """Check that classes are using slots and are weak-referenceable""" x = Selector(text="") weakref.ref(x) assert not hasattr(x, "__dict__"), ( f"{x.__class__.__name__} does not use __slots__" ) def test_selector_bad_args(self): with pytest.raises(ValueError, match="received both response and text"): Selector(TextResponse(url="http://example.com", body=b""), text="") @pytest.mark.skipif(not PARSEL_18_PLUS, reason="parsel < 1.8 doesn't support jmespath") class TestJMESPath: def test_json_has_html(self) -> None: """Sometimes the information is returned in a json wrapper""" body = """ { "content": [ { "name": "A", "value": "a" }, { "name": { "age": 18 }, "value": "b" }, { "name": "C", "value": "c" }, { "name": "D", "value": "
d
" } ], "html": "
a
b
c
def
" } """ resp = TextResponse(url="http://example.com", body=body, encoding="utf-8") assert ( resp.jmespath("html").get() == "
a
b
c
def
" ) assert resp.jmespath("html").xpath("//div/a/text()").getall() == ["a", "b", "d"] assert resp.jmespath("html").css("div > b").getall() == ["f"] assert resp.jmespath("content").jmespath("name.age").get() == "18" def test_html_has_json(self) -> None: body = """

Information

{ "user": [ { "name": "A", "age": 18 }, { "name": "B", "age": 32 }, { "name": "C", "age": 22 }, { "name": "D", "age": 25 } ], "total": 4, "status": "ok" }
""" resp = TextResponse(url="http://example.com", body=body, encoding="utf-8") assert resp.xpath("//div/content/text()").jmespath("user[*].name").getall() == [ "A", "B", "C", "D", ] assert resp.xpath("//div/content").jmespath("user[*].name").getall() == [ "A", "B", "C", "D", ] assert resp.xpath("//div/content").jmespath("total").get() == "4" def test_jmestpath_with_re(self) -> None: body = """

Information

{ "user": [ { "name": "A", "age": 18 }, { "name": "B", "age": 32 }, { "name": "C", "age": 22 }, { "name": "D", "age": 25 } ], "total": 4, "status": "ok" }
""" resp = TextResponse(url="http://example.com", body=body, encoding="utf-8") assert resp.xpath("//div/content/text()").jmespath("user[*].name").re( r"(\w+)" ) == ["A", "B", "C", "D"] assert resp.xpath("//div/content").jmespath("user[*].name").re(r"(\w+)") == [ "A", "B", "C", "D", ] assert resp.xpath("//div/content").jmespath("unavailable").re(r"(\d+)") == [] assert ( resp.xpath("//div/content").jmespath("unavailable").re_first(r"(\d+)") is None ) assert resp.xpath("//div/content").jmespath("user[*].age.to_string(@)").re( r"(\d+)" ) == ["18", "32", "22", "25"] @pytest.mark.skipif(PARSEL_18_PLUS, reason="parsel >= 1.8 supports jmespath") def test_jmespath_not_available() -> None: body = """ { "website": {"name": "Example"} } """ resp = TextResponse(url="http://example.com", body=body, encoding="utf-8") with pytest.raises(AttributeError): resp.jmespath("website.name").get()