diff --git a/scrapy/tests/test_utils_response.py b/scrapy/tests/test_utils_response.py
index a317cd732..71c14569e 100644
--- a/scrapy/tests/test_utils_response.py
+++ b/scrapy/tests/test_utils_response.py
@@ -29,13 +29,29 @@ class ResponseUtilsTest(unittest.TestCase):
self.assertTrue(isinstance(body_or_str(u'text', unicode=True), unicode))
def test_get_base_url(self):
- response = HtmlResponse(url='http://example.org', body="""\
+ response = HtmlResponse(url='https://example.org', body="""\
\
Dummy\
blahablsdfsal&\
""")
self.assertEqual(get_base_url(response), 'http://example.org/something')
+ # relative url with absolute path
+ response = HtmlResponse(url='https://example.org', body="""\
+ \
+ Dummy\
+ blahablsdfsal&\
+ """)
+ self.assertEqual(get_base_url(response), 'https://example.org/absolutepath')
+
+ # no scheme url
+ response = HtmlResponse(url='https://example.org', body="""\
+ \
+ Dummy\
+ blahablsdfsal&\
+ """)
+ self.assertEqual(get_base_url(response), 'https://noscheme.com/path')
+
def test_get_meta_refresh(self):
body = """
diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py
index 15b2afbab..ca258ef44 100644
--- a/scrapy/utils/response.py
+++ b/scrapy/utils/response.py
@@ -33,7 +33,7 @@ def get_base_url(response):
""" Return the base url of the given response used to resolve relative links. """
if response not in _baseurl_cache:
match = BASEURL_RE.search(response.body_as_unicode()[0:4096])
- _baseurl_cache[response] = match.group(1) if match else response.url
+ _baseurl_cache[response] = urljoin_rfc(response.url, match.group(1)) if match else response.url
return _baseurl_cache[response]
META_REFRESH_RE = re.compile(ur']*http-equiv[^>]*refresh[^>]*content\s*=\s*(?P["\'])(?P\d+)\s*;\s*url=(?P.*?)(?P=quote)', \