From 628a3afbbda919ef1fbe7a11772842564135123a Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 23 Jul 2026 13:38:56 +0500 Subject: [PATCH] Remove deprecated xmliter(), deprecate re_rsearch() (#7765) --- docs/news.rst | 2 +- scrapy/utils/iterators.py | 60 ---------------------------- scrapy/utils/python.py | 8 +++- tests/test_utils_iterators.py | 75 ++++++++--------------------------- 4 files changed, 24 insertions(+), 121 deletions(-) diff --git a/docs/news.rst b/docs/news.rst index 4ebab3c6a..8f8477eaa 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -5155,7 +5155,7 @@ Bug fixes * The system file mode creation mask no longer affects the permissions of files generated using the :command:`startproject` command (:issue:`4722`) -* :func:`scrapy.utils.iterators.xmliter` now supports namespaced node names +* ``scrapy.utils.iterators.xmliter`` now supports namespaced node names (:issue:`861`, :issue:`4746`) * :class:`~scrapy.Request` objects can now have ``about:`` URLs, which can diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index 9bb64c021..540fe407c 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -2,7 +2,6 @@ from __future__ import annotations import csv import logging -import re from io import StringIO from typing import TYPE_CHECKING, Any, Literal, cast, overload from warnings import warn @@ -12,7 +11,6 @@ from lxml import etree from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response, TextResponse from scrapy.selector import Selector -from scrapy.utils.python import re_rsearch if TYPE_CHECKING: from collections.abc import Callable, Iterator @@ -20,64 +18,6 @@ if TYPE_CHECKING: logger = logging.getLogger(__name__) -def xmliter(obj: Response | str | bytes, nodename: str) -> Iterator[Selector]: - """Return a iterator of Selector's over all nodes of a XML document, - given the name of the node to iterate. Useful for parsing XML feeds. - - obj can be: - - a Response object - - a unicode string - - a string encoded as utf-8 - """ - warn( - ( - "xmliter is deprecated and its use strongly discouraged because " - "it is vulnerable to ReDoS attacks. Use xmliter_lxml instead. See " - "https://github.com/scrapy/scrapy/security/advisories/GHSA-cc65-xxvf-f7r9" - ), - ScrapyDeprecationWarning, - stacklevel=2, - ) - - nodename_patt = re.escape(nodename) - - DOCUMENT_HEADER_RE = re.compile(r"<\?xml[^>]+>\s*", re.DOTALL) - HEADER_END_RE = re.compile(rf"<\s*/{nodename_patt}\s*>", re.DOTALL) - END_TAG_RE = re.compile(r"<\s*/([^\s>]+)\s*>", re.DOTALL) - NAMESPACE_RE = re.compile(r"((xmlns[:A-Za-z]*)=[^>\s]+)", re.DOTALL) - text = _body_or_str(obj) - - document_header_match = re.search(DOCUMENT_HEADER_RE, text) - document_header = ( - document_header_match.group().strip() if document_header_match else "" - ) - header_end_idx = re_rsearch(HEADER_END_RE, text) - header_end = text[header_end_idx[1] :].strip() if header_end_idx else "" - namespaces: dict[str, str] = {} - if header_end: - for tagname in reversed(re.findall(END_TAG_RE, header_end)): - assert header_end_idx - tag = re.search( - rf"<\s*{tagname}.*?xmlns[:=][^>]*>", - text[: header_end_idx[1]], - re.DOTALL, - ) - if tag: - for x in re.findall(NAMESPACE_RE, tag.group()): - namespaces[x[1]] = x[0] - - r = re.compile(rf"<{nodename_patt}[\s>].*?", re.DOTALL) - for match in r.finditer(text): - nodetext = ( - document_header - + match.group().replace( - nodename, f"{nodename} {' '.join(namespaces.values())}", 1 - ) - + header_end - ) - yield Selector(text=nodetext, type="xml") - - def xmliter_lxml( obj: Response | str | bytes, nodename: str, diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 00cf818b7..8a7517c1d 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -113,7 +113,7 @@ def _chunk_iter(text: str, chunk_size: int) -> Iterable[tuple[str, int]]: def re_rsearch( pattern: str | Pattern[str], text: str, chunk_size: int = 1024 -) -> tuple[int, int] | None: +) -> tuple[int, int] | None: # pragma: no cover """ This function does a reverse search in a text using a regular expression given in the attribute 'pattern'. @@ -127,6 +127,12 @@ def re_rsearch( the start position of the match, and the ending (regarding the entire text). """ + warnings.warn( + "re_rsearch() is deprecated and will be removed in a future Scrapy version.", + category=ScrapyDeprecationWarning, + stacklevel=2, + ) + if isinstance(pattern, str): pattern = re.compile(pattern) diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index 4fb18305e..d431b8017 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -1,28 +1,13 @@ from __future__ import annotations -from abc import ABC, abstractmethod -from typing import TYPE_CHECKING, Any - import pytest -from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response, TextResponse, XmlResponse -from scrapy.utils.iterators import _body_or_str, csviter, xmliter, xmliter_lxml +from scrapy.utils.iterators import _body_or_str, csviter, xmliter_lxml from tests import get_testdata -if TYPE_CHECKING: - from collections.abc import Iterator - - from scrapy import Selector - - -class TestXmliterBase(ABC): - @abstractmethod - def xmliter( - self, obj: Response | str | bytes, nodename: str, *args: Any - ) -> Iterator[Selector]: - raise NotImplementedError +class TestXmliter: def test_xmliter(self): body = b""" @@ -46,7 +31,7 @@ class TestXmliterBase(ABC): x.xpath("name/text()").getall(), x.xpath("./type/text()").getall(), ) - for x in self.xmliter(response, "product") + for x in xmliter_lxml(response, "product") ] assert attrs == [ @@ -63,7 +48,7 @@ class TestXmliterBase(ABC): """ response = XmlResponse(url="http://example.com", body=body) nodenames = [ - e.xpath("name()").getall() for e in self.xmliter(response, "matchme...") + e.xpath("name()").getall() for e in xmliter_lxml(response, "matchme...") ] assert nodenames == [["matchme..."]] @@ -117,7 +102,7 @@ class TestXmliterBase(ABC): x.xpath("./skammstafanir/stuttskammstöfun/text()").getall(), x.xpath("./tímabil/fyrstaþing/text()").getall(), ) - for x in self.xmliter(r, "þingflokkur") + for x in xmliter_lxml(r, "þingflokkur") ] assert attrs == [ @@ -132,7 +117,7 @@ class TestXmliterBase(ABC): "onetwo" ) - assert [x.xpath("text()").getall() for x in self.xmliter(body, "product")] == [ + assert [x.xpath("text()").getall() for x in xmliter_lxml(body, "product")] == [ ["one"], ["two"], ] @@ -157,7 +142,7 @@ class TestXmliterBase(ABC): """ response = XmlResponse(url="http://mydummycompany.com", body=body) - my_iter = self.xmliter(response, "item") + my_iter = xmliter_lxml(response, "item") node = next(my_iter) node.register_namespace("g", "http://base.google.com/ns/1.0") assert node.xpath("title/text()").getall() == ["Item 1"] @@ -194,7 +179,7 @@ class TestXmliterBase(ABC): """ response = XmlResponse(url="http://mydummycompany.com", body=body) - my_iter = self.xmliter(response, "g:image_link") + my_iter = xmliter_lxml(response, "g:image_link") node = next(my_iter) node.register_namespace("g", "http://base.google.com/ns/1.0") assert node.xpath("text()").extract() == [ @@ -221,7 +206,7 @@ class TestXmliterBase(ABC): """ response = XmlResponse(url="http://mydummycompany.com", body=body) - my_iter = self.xmliter(response, "g:link_image") + my_iter = xmliter_lxml(response, "g:link_image") with pytest.raises(StopIteration): next(my_iter) @@ -231,14 +216,14 @@ class TestXmliterBase(ABC): "onetwo" ) - my_iter = self.xmliter(body, "product") + my_iter = xmliter_lxml(body, "product") next(my_iter) next(my_iter) with pytest.raises(StopIteration): next(my_iter) def test_xmliter_objtype_exception(self): - i = self.xmliter(42, "product") # type: ignore[arg-type] + i = xmliter_lxml(42, "product") # type: ignore[arg-type] with pytest.raises(TypeError): next(i) @@ -251,38 +236,10 @@ class TestXmliterBase(ABC): ) response = XmlResponse("http://www.example.com", body=body) assert ( - next(self.xmliter(response, "item")).get() + next(xmliter_lxml(response, "item")).get() == "Some Turkish Characters \xd6\xc7\u015e\u0130\u011e\xdc \xfc\u011f\u0131\u015f\xe7\xf6" ) - -@pytest.mark.filterwarnings("ignore::scrapy.exceptions.ScrapyDeprecationWarning") -class TestXmliter(TestXmliterBase): - def xmliter( - self, obj: Response | str | bytes, nodename: str, *args: Any - ) -> Iterator[Selector]: - return xmliter(obj, nodename) - - def test_deprecation(self): - body = b""" - - - - - """ - with pytest.warns( - ScrapyDeprecationWarning, - match="xmliter", - ): - next(self.xmliter(body, "product")) - - -class TestLxmlXmliter(TestXmliterBase): - def xmliter( - self, obj: Response | str | bytes, nodename: str, *args: Any - ) -> Iterator[Selector]: - return xmliter_lxml(obj, nodename, *args) - def test_xmliter_iterate_namespace(self): body = b""" @@ -303,10 +260,10 @@ class TestLxmlXmliter(TestXmliterBase): """ response = XmlResponse(url="http://mydummycompany.com", body=body) - no_namespace_iter = self.xmliter(response, "image_link") + no_namespace_iter = xmliter_lxml(response, "image_link") assert len(list(no_namespace_iter)) == 0 - namespace_iter = self.xmliter( + namespace_iter = xmliter_lxml( response, "image_link", "http://base.google.com/ns/1.0" ) node = next(namespace_iter) @@ -338,14 +295,14 @@ class TestLxmlXmliter(TestXmliterBase): """ response = XmlResponse(url="http://mydummycompany.com", body=body) - my_iter = self.xmliter(response, "table", "http://www.w3.org/TR/html4/", "h") + my_iter = xmliter_lxml(response, "table", "http://www.w3.org/TR/html4/", "h") node = next(my_iter) assert len(node.xpath("h:tr/h:td").getall()) == 2 assert node.xpath("h:tr/h:td[1]/text()").getall() == ["Apples"] assert node.xpath("h:tr/h:td[2]/text()").getall() == ["Bananas"] - my_iter = self.xmliter( + my_iter = xmliter_lxml( response, "table", "http://www.w3schools.com/furniture", "f" )