From 7f4f98fd38d3fdf6b45a9d0289df0cbb48bcd22b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 30 Sep 2019 18:22:28 +0200 Subject: [PATCH 1/6] Provide complete API documentation coverage of scrapy.linkextractors --- docs/conf.py | 4 +++ docs/topics/link-extractors.rst | 45 ++++++++++++------------------- scrapy/linkextractors/__init__.py | 11 ++++++++ scrapy/linkextractors/lxmlhtml.py | 8 ++++++ tests/test_linkextractors.py | 30 +++++++++++++++++++++ 5 files changed, 70 insertions(+), 28 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 34dd5bcb7..5ba6b8d43 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -265,6 +265,10 @@ coverage_ignore_pyobjects = [ # Never documented before, and deprecated now. r'^scrapy\.item\.DictItem$', + r'^scrapy\.linkextractors\.FilteringLinkExtractor$', + + # Implementation detail of LxmlLinkExtractor + r'^scrapy\.linkextractors\.lxmlhtml\.LxmlParserLinkExtractor', ] diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index 713a94e10..f9936a498 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -4,46 +4,33 @@ Link Extractors =============== -Link extractors are objects whose only purpose is to extract links from web -pages (:class:`scrapy.http.Response` objects) which will be eventually -followed. +A link extractor is an object that extracts links from responses. -There is ``scrapy.linkextractors.LinkExtractor`` available -in Scrapy, but you can create your own custom Link Extractors to suit your -needs by implementing a simple interface. - -The only public method that every link extractor has is ``extract_links``, -which receives a :class:`~scrapy.http.Response` object and returns a list -of :class:`scrapy.link.Link` objects. Link extractors are meant to be -instantiated once and their ``extract_links`` method called several times -with different responses to extract links to follow. - -Link extractors are used in the :class:`~scrapy.spiders.CrawlSpider` -class (available in Scrapy), through a set of rules, but you can also use it in -your spiders, even if you don't subclass from -:class:`~scrapy.spiders.CrawlSpider`, as its purpose is very simple: to -extract links. +The constructor of :class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` +takes settings that determine which links may be extracted. +:class:`LxmlLinkExtractor.extract_links +` returns a +list of matching :class:`scrapy.link.Link` objects from a +:class:`~scrapy.http.Response` object. +Link extractors are used in :class:`~scrapy.spiders.CrawlSpider` spiders +through a set of :class:`~scrapy.spiders.Rule` objects. You can also use link +extractors in regular spiders. .. _topics-link-extractors-ref: -Built-in link extractors reference -================================== +Link extractor reference +======================== .. module:: scrapy.linkextractors :synopsis: Link extractors classes -Link extractors classes bundled with Scrapy are provided in the -:mod:`scrapy.linkextractors` module. - -The default link extractor is ``LinkExtractor``, which is the same as -:class:`~.LxmlLinkExtractor`:: +The link extractor class is +:class:`scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`. For convenience it +can also be imported as ``scrapy.linkextractors.LinkExtractor``:: from scrapy.linkextractors import LinkExtractor -There used to be other link extractor classes in previous Scrapy versions, -but they are deprecated now. - LxmlLinkExtractor ----------------- @@ -152,4 +139,6 @@ LxmlLinkExtractor from elements or attributes which allow leading/trailing whitespaces). :type strip: boolean + .. automethod:: extract_links + .. _scrapy.linkextractors: https://github.com/scrapy/scrapy/blob/master/scrapy/linkextractors/__init__.py diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index ebf3cd7d8..ca80dc339 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -6,11 +6,13 @@ This package contains a collection of Link Extractors. For more info see docs/topics/link-extractors.rst """ import re +from warnings import warn from six.moves.urllib.parse import urlparse from parsel.csstranslator import HTMLTranslator from w3lib.url import canonicalize_url +from scrapy.utils.deprecate import ScrapyDeprecationWarning from scrapy.utils.misc import arg_to_iter from scrapy.utils.url import ( url_is_from_any_domain, url_has_any_extension, @@ -49,6 +51,15 @@ class FilteringLinkExtractor(object): _csstranslator = HTMLTranslator() + def __new__(cls, *args, **kwargs): + from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor + if (issubclass(cls, FilteringLinkExtractor) and + not issubclass(cls, LxmlLinkExtractor)): + warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' + 'please use scrapy.linkextractors.LinkExtractor instead', + ScrapyDeprecationWarning, stacklevel=2) + return super(FilteringLinkExtractor, cls).__new__(cls) + def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text): diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index 8f6f93a44..41091ba23 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -117,6 +117,14 @@ class LxmlLinkExtractor(FilteringLinkExtractor): restrict_text=restrict_text) def extract_links(self, response): + """Returns a list of :class:`~scrapy.link.Link` objects from the + specified :class:`response `. + + Only links that match the settings passed to the link extractor + constructor are returned. + + Duplicate links are omitted. + """ base_url = get_base_url(response) if self.restrict_xpaths: docs = [subdoc diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index d96e259f6..ea6db28c0 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -1,10 +1,13 @@ import re import unittest +from warnings import catch_warnings import pytest +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import HtmlResponse, XmlResponse from scrapy.link import Link +from scrapy.linkextractors import FilteringLinkExtractor from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor from tests import get_testdata @@ -506,3 +509,30 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): @pytest.mark.xfail def test_restrict_xpaths_with_html_entities(self): super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities() + + def test_filteringlinkextractor_deprecation_warning(self): + """Make sure the FilteringLinkExtractor deprecation warning is not + issued for LxmlLinkExtractor""" + with catch_warnings(record=True) as warnings: + extractor = LxmlLinkExtractor() + self.assertEqual(len(warnings), 0) + class SubclassedItem(LxmlLinkExtractor): + pass + subclassed_extractor = SubclassedItem() + self.assertEqual(len(warnings), 0) + + +class FilteringLinkExtractorTest(unittest.TestCase): + + def test_deprecation_warning(self): + args = [None]*10 + with catch_warnings(record=True) as warnings: + extractor = FilteringLinkExtractor(*args) + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) + with catch_warnings(record=True) as warnings: + class SubclassedFilteringLinkExtractor(FilteringLinkExtractor): + pass + subclassed_extractor = SubclassedFilteringLinkExtractor(*args) + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) From 0fbd1ff4a91c68e552a3f919824c60437fc9a141 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 21 Oct 2019 14:06:45 +0200 Subject: [PATCH 2/6] =?UTF-8?q?constructor=20=E2=86=92=20=5F=5Finit=5F=5F?= =?UTF-8?q?=20method?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/topics/link-extractors.rst | 6 +++--- scrapy/linkextractors/lxmlhtml.py | 4 ++-- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index f9936a498..2119cb8f8 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -6,9 +6,9 @@ Link Extractors A link extractor is an object that extracts links from responses. -The constructor of :class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` -takes settings that determine which links may be extracted. -:class:`LxmlLinkExtractor.extract_links +The ``__init__`` method of +:class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` takes settings that +determine which links may be extracted. :class:`LxmlLinkExtractor.extract_links ` returns a list of matching :class:`scrapy.link.Link` objects from a :class:`~scrapy.http.Response` object. diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index 41091ba23..37003720f 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -120,8 +120,8 @@ class LxmlLinkExtractor(FilteringLinkExtractor): """Returns a list of :class:`~scrapy.link.Link` objects from the specified :class:`response `. - Only links that match the settings passed to the link extractor - constructor are returned. + Only links that match the settings passed to the ``__init__`` method of + the link extractor are returned. Duplicate links are omitted. """ From a4ef9750f9058fbe1041bad0fb28f39c693e5659 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Fri, 13 Dec 2019 14:32:06 +0100 Subject: [PATCH 3/6] Fix Flake8-reported issues --- pytest.ini | 2 +- tests/test_linkextractors.py | 12 +++++++----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/pytest.ini b/pytest.ini index 33c34b8e8..02014d1a5 100644 --- a/pytest.ini +++ b/pytest.ini @@ -88,7 +88,7 @@ flake8-ignore = scrapy/http/response/__init__.py E501 E128 W293 W291 scrapy/http/response/text.py E501 W293 E128 E124 # scrapy/linkextractors - scrapy/linkextractors/__init__.py E731 E502 E501 E402 + scrapy/linkextractors/__init__.py E731 E502 E501 E402 W504 scrapy/linkextractors/lxmlhtml.py E501 E731 E226 # scrapy/loader scrapy/loader/__init__.py E501 E502 E128 diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index ebe497913..0ffeaecc3 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -514,25 +514,27 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): """Make sure the FilteringLinkExtractor deprecation warning is not issued for LxmlLinkExtractor""" with catch_warnings(record=True) as warnings: - extractor = LxmlLinkExtractor() + LxmlLinkExtractor() self.assertEqual(len(warnings), 0) + class SubclassedItem(LxmlLinkExtractor): pass - subclassed_extractor = SubclassedItem() + + SubclassedItem() self.assertEqual(len(warnings), 0) class FilteringLinkExtractorTest(unittest.TestCase): def test_deprecation_warning(self): - args = [None]*10 + args = [None] * 10 with catch_warnings(record=True) as warnings: - extractor = FilteringLinkExtractor(*args) + FilteringLinkExtractor(*args) self.assertEqual(len(warnings), 1) self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) with catch_warnings(record=True) as warnings: class SubclassedFilteringLinkExtractor(FilteringLinkExtractor): pass - subclassed_extractor = SubclassedFilteringLinkExtractor(*args) + SubclassedFilteringLinkExtractor(*args) self.assertEqual(len(warnings), 1) self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) From ee9881d2704798c9cd61b6da503bb0694227c58c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 18 Dec 2019 12:08:34 +0100 Subject: [PATCH 4/6] Improve FilteringLinkExtractor.__new__ --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index a510fef70..7254bd79c 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -61,7 +61,7 @@ class FilteringLinkExtractor(object): warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' 'please use scrapy.linkextractors.LinkExtractor instead', ScrapyDeprecationWarning, stacklevel=2) - return super(FilteringLinkExtractor, cls).__new__(cls) + return super().__new__(cls, *args, **kwargs) def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text): From 174769a3f08fcd84eaec8a88217a05f8ebc3f2cb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 18 Dec 2019 12:09:03 +0100 Subject: [PATCH 5/6] Use a better name for the LxmlLinkExtractor subclassing test --- tests/test_linkextractors.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index 0ffeaecc3..cfd4c6b85 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -517,10 +517,10 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): LxmlLinkExtractor() self.assertEqual(len(warnings), 0) - class SubclassedItem(LxmlLinkExtractor): + class SubclassedLxmlLinkExtractor(LxmlLinkExtractor): pass - SubclassedItem() + SubclassedLxmlLinkExtractor() self.assertEqual(len(warnings), 0) From e22c0c27d9d33383f4ac18a4142ccffe1d9a0d3b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 19 Dec 2019 12:15:54 +0100 Subject: [PATCH 6/6] Revert "Improve FilteringLinkExtractor.__new__" This reverts commit ee9881d2704798c9cd61b6da503bb0694227c58c. --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index d0d34035d..bdeab3a75 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -60,7 +60,7 @@ class FilteringLinkExtractor(object): warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' 'please use scrapy.linkextractors.LinkExtractor instead', ScrapyDeprecationWarning, stacklevel=2) - return super().__new__(cls, *args, **kwargs) + return super(FilteringLinkExtractor, cls).__new__(cls) def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text):