From dd10cb8e9a982fe3d311078d6e1207596e272717 Mon Sep 17 00:00:00 2001 From: Adrian Date: Sun, 5 Jul 2026 13:46:22 +0200 Subject: [PATCH] LxmlLinkExtractor: add deny_attrs and deny_tags (#7679) --- docs/topics/link-extractors.rst | 103 +--------------------- scrapy/linkextractors/lxmlhtml.py | 142 +++++++++++++++++++++++++++++- tests/test_linkextractors.py | 53 +++++++++++ 3 files changed, 194 insertions(+), 104 deletions(-) diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index 613e175da..3fc896507 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -47,108 +47,7 @@ LxmlLinkExtractor :synopsis: lxml's HTMLParser-based link extractors -.. class:: LxmlLinkExtractor(allow=(), deny=(), allow_domains=(), deny_domains=(), deny_extensions=None, restrict_xpaths=(), restrict_css=(), tags=('a', 'area'), attrs=('href',), canonicalize=False, unique=True, process_value=None, strip=True) - - LxmlLinkExtractor is the recommended link extractor with handy filtering - options. It is implemented using lxml's robust HTMLParser. - - :param allow: a single regular expression (or list of regular expressions) - that the (absolute) urls must match in order to be extracted. If not - given (or empty), it will match all links. - :type allow: str or list - - :param deny: a single regular expression (or list of regular expressions) - that the (absolute) urls must match in order to be excluded (i.e. not - extracted). It has precedence over the ``allow`` parameter. If not - given (or empty) it won't exclude any links. - :type deny: str or list - - :param allow_domains: a single value or a list of string containing - domains which will be considered for extracting the links - :type allow_domains: str or list - - :param deny_domains: a single value or a list of strings containing - domains which won't be considered for extracting the links - :type deny_domains: str or list - - :param deny_extensions: a single value or list of strings containing - extensions that should be ignored when extracting links. - If not given, it will default to - :data:`scrapy.linkextractors.IGNORED_EXTENSIONS`. - - :type deny_extensions: list - - :param restrict_xpaths: is an XPath (or list of XPath's) which defines - regions inside the response where links should be extracted from. - If given, only the text selected by those XPath will be scanned for - links. - :type restrict_xpaths: str or list - - :param restrict_css: a CSS selector (or list of selectors) which defines - regions inside the response where links should be extracted from. - Has the same behaviour as ``restrict_xpaths``. - :type restrict_css: str or list - - :param restrict_text: a single regular expression (or list of regular expressions) - that the link's text must match in order to be extracted. If not - given (or empty), it will match all links. If a list of regular expressions is - given, the link will be extracted if it matches at least one. - :type restrict_text: str or list - - :param tags: a tag or a list of tags to consider when extracting links. - Defaults to ``('a', 'area')``. - :type tags: str or list - - :param attrs: an attribute or list of attributes which should be considered when looking - for links to extract (only for those tags specified in the ``tags`` - parameter). Defaults to ``('href',)`` - :type attrs: list - - :param canonicalize: canonicalize each extracted url (using - w3lib.url.canonicalize_url). Defaults to ``False``. - Note that canonicalize_url is meant for duplicate checking; - it can change the URL visible at server side, so the response can be - different for requests with canonicalized and raw URLs. If you're - using LinkExtractor to follow links it is more robust to - keep the default ``canonicalize=False``. - :type canonicalize: bool - - :param unique: whether duplicate filtering should be applied to extracted - links. - :type unique: bool - - :param process_value: a function which receives each value extracted from - the tag and attributes scanned and can modify the value and return a - new one, or return ``None`` to ignore the link altogether. If not - given, ``process_value`` defaults to ``lambda x: x``. - - .. highlight:: html - - For example, to extract links from this code:: - - Link text - - .. highlight:: python - - You can use the following function in ``process_value``: - - .. code-block:: python - - def process_value(value): - m = re.search(r"javascript:goToPage\('(.*?)'", value) - if m: - return m.group(1) - - :type process_value: collections.abc.Callable - - :param strip: whether to strip whitespaces from extracted attributes. - According to HTML5 standard, leading and trailing whitespaces - must be stripped from ``href`` attributes of ````, ```` - and many other elements, ``src`` attribute of ````, ``