From f576b3ffeeb86ceb88631e14dd56ea5b96f53e35 Mon Sep 17 00:00:00 2001 From: Mikhail Korobov Date: Sat, 25 Jul 2015 00:40:45 +0200 Subject: [PATCH] [tmp] improve python 3 support for scrapy.utils.url --- scrapy/utils/python.py | 9 +++++++ scrapy/utils/url.py | 22 ++++++++++------- tests/test_utils_url.py | 53 +++++++++++++++++++++++++++-------------- 3 files changed, 58 insertions(+), 26 deletions(-) diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 57016811f..94ee8a557 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -120,6 +120,15 @@ def to_bytes(text, encoding=None, errors='strict'): return text.encode(encoding, errors) +def to_native_str(text, encoding=None, errors='strict'): + """ Return str representation of `text` + (bytes in Python 2.x and unicode in Python 3.x). """ + if six.PY2: + return to_bytes(text, encoding, errors) + else: + return to_unicode(text, encoding, errors) + + def re_rsearch(pattern, text, chunk_size=1024): """ This function does a reverse search in a text using a regular expression diff --git a/scrapy/utils/url.py b/scrapy/utils/url.py index 8a8c56814..99f350361 100644 --- a/scrapy/utils/url.py +++ b/scrapy/utils/url.py @@ -10,19 +10,20 @@ from six.moves.urllib.parse import (ParseResult, urlunparse, urldefrag, urlparse, parse_qsl, urlencode, unquote) -# scrapy.utils.url was moved to w3lib.url and import * ensures this move doesn't break old code +# scrapy.utils.url was moved to w3lib.url and import * ensures this +# move doesn't break old code from w3lib.url import * -from scrapy.utils.python import to_bytes +from w3lib.url import _safe_chars +from scrapy.utils.python import to_native_str def url_is_from_any_domain(url, domains): """Return True if the url belongs to any of the given domains""" host = parse_url(url).netloc.lower() - - if host: - return any(((host == d.lower()) or (host.endswith('.%s' % d.lower())) for d in domains)) - else: + if not host: return False + domains = [d.lower() for d in domains] + return any((host == d) or (host.endswith('.%s' % d)) for d in domains) def url_is_from_spider(url, spider): @@ -36,7 +37,7 @@ def url_has_any_extension(url, extensions): def canonicalize_url(url, keep_blank_values=True, keep_fragments=False, - encoding=None): + encoding=None): """Canonicalize the given url by applying the following procedures: - sort query arguments, first by key, then by value @@ -57,6 +58,11 @@ def canonicalize_url(url, keep_blank_values=True, keep_fragments=False, keyvals = parse_qsl(query, keep_blank_values) keyvals.sort() query = urlencode(keyvals) + + # XXX: copied from w3lib.url.safe_url_string to add encoding argument + # path = to_native_str(path, encoding) + # path = moves.urllib.parse.quote(path, _safe_chars, encoding='latin1') or '/' + path = safe_url_string(_unquotepath(path)) or '/' fragment = '' if not keep_fragments else fragment return urlunparse((scheme, netloc.lower(), path, params, query, fragment)) @@ -74,7 +80,7 @@ def parse_url(url, encoding=None): """ if isinstance(url, ParseResult): return url - return urlparse(to_bytes(url, encoding)) + return urlparse(to_native_str(url, encoding)) def escape_ajax(url): diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 860c76bae..7bf0e5b4a 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -1,7 +1,10 @@ +# -*- coding: utf-8 -*- import unittest +import six from scrapy.spiders import Spider -from scrapy.utils.url import url_is_from_any_domain, url_is_from_spider, canonicalize_url +from scrapy.utils.url import (url_is_from_any_domain, url_is_from_spider, + canonicalize_url) __doctests__ = ['scrapy.utils.url'] @@ -70,18 +73,23 @@ class UrlUtilsTest(unittest.TestCase): self.assertTrue(url_is_from_spider('http://www.example.net/some/page.html', MySpider)) self.assertFalse(url_is_from_spider('http://www.example.us/some/page.html', MySpider)) + +class CanonicalizeUrlTest(unittest.TestCase): + def test_canonicalize_url(self): # simplest case self.assertEqual(canonicalize_url("http://www.example.com/"), "http://www.example.com/") - # always return a str + def test_return_str(self): assert isinstance(canonicalize_url(u"http://www.example.com"), str) + assert isinstance(canonicalize_url(b"http://www.example.com"), str) - # append missing path + def test_append_missing_path(self): self.assertEqual(canonicalize_url("http://www.example.com"), "http://www.example.com/") - # typical usage + + def test_typical_usage(self): self.assertEqual(canonicalize_url("http://www.example.com/do?a=1&b=2&c=3"), "http://www.example.com/do?a=1&b=2&c=3") self.assertEqual(canonicalize_url("http://www.example.com/do?c=1&b=2&a=3"), @@ -89,11 +97,11 @@ class UrlUtilsTest(unittest.TestCase): self.assertEqual(canonicalize_url("http://www.example.com/do?&a=1"), "http://www.example.com/do?a=1") - # sorting by argument values + def test_sorting(self): self.assertEqual(canonicalize_url("http://www.example.com/do?c=3&b=5&b=2&a=50"), "http://www.example.com/do?a=50&b=2&b=5&c=3") - # using keep_blank_values + def test_keep_blank_values(self): self.assertEqual(canonicalize_url("http://www.example.com/do?b=&a=2", keep_blank_values=False), "http://www.example.com/do?a=2") self.assertEqual(canonicalize_url("http://www.example.com/do?b=&a=2"), @@ -106,7 +114,7 @@ class UrlUtilsTest(unittest.TestCase): self.assertEqual(canonicalize_url(u'http://www.example.com/do?1750,4'), 'http://www.example.com/do?1750%2C4=') - # spaces + def test_spaces(self): self.assertEqual(canonicalize_url("http://www.example.com/do?q=a space&a=1"), "http://www.example.com/do?a=1&q=a+space") self.assertEqual(canonicalize_url("http://www.example.com/do?q=a+space&a=1"), @@ -114,43 +122,52 @@ class UrlUtilsTest(unittest.TestCase): self.assertEqual(canonicalize_url("http://www.example.com/do?q=a%20space&a=1"), "http://www.example.com/do?a=1&q=a+space") - # normalize percent-encoding case (in paths) + @unittest.skipUnless(six.PY2, "TODO") + def test_normalize_percent_encoding_in_paths(self): self.assertEqual(canonicalize_url("http://www.example.com/a%a3do"), "http://www.example.com/a%A3do"), - # normalize percent-encoding case (in query arguments) + + @unittest.skipUnless(six.PY2, "TODO") + def test_normalize_percent_encoding_in_query_arguments(self): self.assertEqual(canonicalize_url("http://www.example.com/do?k=b%a3"), "http://www.example.com/do?k=b%A3") - # non-ASCII percent-encoding in paths + def test_non_ascii_percent_encoding_in_paths(self): self.assertEqual(canonicalize_url("http://www.example.com/a do?a=1"), "http://www.example.com/a%20do?a=1"), self.assertEqual(canonicalize_url("http://www.example.com/a %20do?a=1"), "http://www.example.com/a%20%20do?a=1"), - self.assertEqual(canonicalize_url("http://www.example.com/a do\xc2\xa3.html?a=1"), + self.assertEqual(canonicalize_url(u"http://www.example.com/a do£.html?a=1"), "http://www.example.com/a%20do%C2%A3.html?a=1") - # non-ASCII percent-encoding in query arguments + self.assertEqual(canonicalize_url(b"http://www.example.com/a do\xc2\xa3.html?a=1"), + "http://www.example.com/a%20do%C2%A3.html?a=1") + + def test_non_ascii_percent_encoding_in_query_arguments(self): self.assertEqual(canonicalize_url(u"http://www.example.com/do?price=\xa3500&a=5&z=3"), u"http://www.example.com/do?a=5&price=%C2%A3500&z=3") - self.assertEqual(canonicalize_url("http://www.example.com/do?price=\xc2\xa3500&a=5&z=3"), + self.assertEqual(canonicalize_url(b"http://www.example.com/do?price=\xc2\xa3500&a=5&z=3"), "http://www.example.com/do?a=5&price=%C2%A3500&z=3") - self.assertEqual(canonicalize_url("http://www.example.com/do?price(\xc2\xa3)=500&a=1"), + self.assertEqual(canonicalize_url(b"http://www.example.com/do?price(\xc2\xa3)=500&a=1"), "http://www.example.com/do?a=1&price%28%C2%A3%29=500") - # urls containing auth and ports + def test_urls_with_auth_and_ports(self): self.assertEqual(canonicalize_url(u"http://user:pass@www.example.com:81/do?now=1"), u"http://user:pass@www.example.com:81/do?now=1") - # remove fragments + def test_remove_fragments(self): self.assertEqual(canonicalize_url(u"http://user:pass@www.example.com/do?a=1#frag"), u"http://user:pass@www.example.com/do?a=1") self.assertEqual(canonicalize_url(u"http://user:pass@www.example.com/do?a=1#frag", keep_fragments=True), u"http://user:pass@www.example.com/do?a=1#frag") + def test_dont_convert_safe_characters(self): # dont convert safe characters to percent encoding representation self.assertEqual(canonicalize_url( "http://www.simplybedrooms.com/White-Bedroom-Furniture/Bedroom-Mirror:-Josephine-Cheval-Mirror.html"), "http://www.simplybedrooms.com/White-Bedroom-Furniture/Bedroom-Mirror:-Josephine-Cheval-Mirror.html") + @unittest.skipUnless(six.PY2, "TODO") + def test_safe_characters_unicode(self): # urllib.quote uses a mapping cache of encoded characters. when parsing # an already percent-encoded url, it will fail if that url was not # percent-encoded as utf-8, that's why canonicalize_url must always @@ -159,11 +176,11 @@ class UrlUtilsTest(unittest.TestCase): self.assertEqual(canonicalize_url(u'http://www.example.com/caf%E9-con-leche.htm'), 'http://www.example.com/caf%E9-con-leche.htm') - # domains are case insensitive + def test_domains_are_case_insensitive(self): self.assertEqual(canonicalize_url("http://www.EXAMPLE.com/"), "http://www.example.com/") - # quoted slash and question sign + def test_quoted_slash_and_question_sign(self): self.assertEqual(canonicalize_url("http://foo.com/AC%2FDC+rocks%3f/?yeah=1"), "http://foo.com/AC%2FDC+rocks%3F/?yeah=1") self.assertEqual(canonicalize_url("http://foo.com/AC%2FDC/"),