From df446d167f6b539c4a79fb64d91cbc1607f30acb Mon Sep 17 00:00:00 2001 From: Mikhail Korobov Date: Tue, 7 Feb 2017 03:11:59 +0500 Subject: [PATCH] fix deprecated link extractors --- scrapy/linkextractors/sgml.py | 18 +++++++++++------- tests/test_linkextractors_deprecated.py | 2 ++ 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/scrapy/linkextractors/sgml.py b/scrapy/linkextractors/sgml.py index 11ff7a261..f4ca4262a 100644 --- a/scrapy/linkextractors/sgml.py +++ b/scrapy/linkextractors/sgml.py @@ -6,7 +6,7 @@ from six.moves.urllib.parse import urljoin import warnings from sgmllib import SGMLParser -from w3lib.url import safe_url_string +from w3lib.url import safe_url_string, canonicalize_url from w3lib.html import strip_html5_whitespace from scrapy.link import Link @@ -20,7 +20,7 @@ from scrapy.exceptions import ScrapyDeprecationWarning class BaseSgmlLinkExtractor(SGMLParser): def __init__(self, tag="a", attr="href", unique=False, process_value=None, - strip=True): + strip=True, canonicalized=False): warnings.warn( "BaseSgmlLinkExtractor is deprecated and will be removed in future releases. " "Please use scrapy.linkextractors.LinkExtractor", @@ -33,6 +33,11 @@ class BaseSgmlLinkExtractor(SGMLParser): self.current_link = None self.unique = unique self.strip = strip + if canonicalized: + self.link_key = lambda link: link.url + else: + self.link_key = lambda link: canonicalize_url(link.url, + keep_fragments=True) def _extract_links(self, response_text, response_url, response_encoding, base_url=None): """ Do the real extraction work """ @@ -61,8 +66,7 @@ class BaseSgmlLinkExtractor(SGMLParser): The subclass should override it if necessary """ - links = unique_list(links, key=lambda link: link.url) if self.unique else links - return links + return unique_list(links, key=self.link_key) if self.unique else links def extract_links(self, response): # wrapper needed to allow to work directly with text @@ -107,10 +111,9 @@ class BaseSgmlLinkExtractor(SGMLParser): class SgmlLinkExtractor(FilteringLinkExtractor): def __init__(self, allow=(), deny=(), allow_domains=(), deny_domains=(), restrict_xpaths=(), - tags=('a', 'area'), attrs=('href',), canonicalize=True, unique=True, + tags=('a', 'area'), attrs=('href',), canonicalize=False, unique=True, process_value=None, deny_extensions=None, restrict_css=(), strip=True): - warnings.warn( "SgmlLinkExtractor is deprecated and will be removed in future releases. " "Please use scrapy.linkextractors.LinkExtractor", @@ -124,7 +127,8 @@ class SgmlLinkExtractor(FilteringLinkExtractor): with warnings.catch_warnings(): warnings.simplefilter('ignore', ScrapyDeprecationWarning) lx = BaseSgmlLinkExtractor(tag=tag_func, attr=attr_func, - unique=unique, process_value=process_value, strip=strip) + unique=unique, process_value=process_value, strip=strip, + canonicalized=canonicalize) super(SgmlLinkExtractor, self).__init__(lx, allow=allow, deny=deny, allow_domains=allow_domains, deny_domains=deny_domains, diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py index fef227aa1..794f85e0f 100644 --- a/tests/test_linkextractors_deprecated.py +++ b/tests/test_linkextractors_deprecated.py @@ -121,6 +121,7 @@ class HtmlParserLinkExtractorTestCase(unittest.TestCase): Link(url='http://example.com/sample2.html', text=u'sample 2'), Link(url='http://example.com/sample3.html', text=u'sample 3 text'), Link(url='http://example.com/sample3.html', text=u'sample 3 repetition'), + Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'), Link(url='http://www.google.com/something', text=u''), Link(url='http://example.com/innertag.html', text=u'inner tag'), Link(url='http://example.com/page%204.html', text=u'href with whitespaces'), @@ -190,6 +191,7 @@ class RegexLinkExtractorTestCase(unittest.TestCase): self.assertEqual(lx.extract_links(self.response), [Link(url='http://example.com/sample2.html', text=u'sample 2'), Link(url='http://example.com/sample3.html', text=u'sample 3 text'), + Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'), Link(url='http://www.google.com/something', text=u''), Link(url='http://example.com/innertag.html', text=u'inner tag'),])