From a41c64bfb97a7bba8ec138345d48b94d8734d27e Mon Sep 17 00:00:00 2001 From: Leonid Amirov Date: Mon, 2 Nov 2015 16:06:21 +0300 Subject: [PATCH] issue GH #1550 - fixed bugs in scrapy.utils.url.add_http_if_no_scheme(): when given URI where scheme is present, but not 'http' the function gave bad result --- scrapy/utils/url.py | 21 ++++++++++++++------- 1 file changed, 14 insertions(+), 7 deletions(-) diff --git a/scrapy/utils/url.py b/scrapy/utils/url.py index 398407a64..3eac5fb3d 100644 --- a/scrapy/utils/url.py +++ b/scrapy/utils/url.py @@ -6,6 +6,7 @@ Some of the functions that used to be imported from this module have been moved to the w3lib.url module. Always import those from there instead. """ import posixpath +from urlparse import urlsplit, urlunsplit from six.moves.urllib.parse import (ParseResult, urlunparse, urldefrag, urlparse, parse_qsl, urlencode, unquote) @@ -114,10 +115,16 @@ def escape_ajax(url): def add_http_if_no_scheme(url): """Add http as the default scheme if it is missing from the url.""" - if url.startswith('//'): - url = 'http:' + url - return url - parser = parse_url(url) - if not parser.scheme or not parser.netloc: - url = 'http://' + url - return url + parts = urlsplit(url) + scheme = parts.scheme or "http" + if parts.netloc: + netloc = parts.netloc + path = parts.path + else: + path_parts = url.split("/", 1) + netloc = path_parts[0] + path = path_parts[1] if len(path_parts) > 1 else "/" + + return urlunsplit(( + scheme, netloc, path, parts.query, parts.fragment + ))