mirror of https://github.com/scrapy/scrapy.git
Moved several functions from scrapy.utils.{http,markup,multipart,response,url} (and their tests) to a new library called 'w3lib'. Scrapy will now depend on w3lib.
This commit is contained in:
parent
ad496eb3b6
commit
b10f4fae35
|
|
@ -2,7 +2,7 @@ Source: scrapy-SUFFIX
|
|||
Section: python
|
||||
Priority: optional
|
||||
Maintainer: Insophia Team <info@insophia.com>
|
||||
Build-Depends: debhelper (>= 7.0.50), python (>=2.5), python-twisted
|
||||
Build-Depends: debhelper (>= 7.0.50), python (>=2.6), python-twisted, python-w3lib
|
||||
Standards-Version: 3.8.4
|
||||
Homepage: http://scrapy.org/
|
||||
|
||||
|
|
|
|||
|
|
@ -1,13 +0,0 @@
|
|||
import unittest
|
||||
from scrapy.utils.http import basic_auth_header
|
||||
|
||||
__doctests__ = ['scrapy.utils.http']
|
||||
|
||||
class UtilsHttpTestCase(unittest.TestCase):
|
||||
|
||||
def test_basic_auth_header(self):
|
||||
self.assertEqual('Basic c29tZXVzZXI6c29tZXBhc3M=',
|
||||
basic_auth_header('someuser', 'somepass'))
|
||||
# Check url unsafe encoded header
|
||||
self.assertEqual('Basic c29tZXVzZXI6QDx5dTk-Jm8_UQ==',
|
||||
basic_auth_header('someuser', '@<yu9>&o?Q'))
|
||||
|
|
@ -1,156 +0,0 @@
|
|||
# -*- coding: utf-8 -*-
|
||||
import unittest
|
||||
|
||||
from scrapy.utils.markup import remove_entities, replace_tags, remove_comments
|
||||
from scrapy.utils.markup import remove_tags_with_content, replace_escape_chars, remove_tags
|
||||
from scrapy.utils.markup import unquote_markup
|
||||
|
||||
class UtilsMarkupTest(unittest.TestCase):
|
||||
|
||||
def test_remove_entities(self):
|
||||
# make sure it always return uncode
|
||||
assert isinstance(remove_entities('no entities'), unicode)
|
||||
assert isinstance(remove_entities('Price: £100!'), unicode)
|
||||
|
||||
# regular conversions
|
||||
self.assertEqual(remove_entities(u'As low as £100!'),
|
||||
u'As low as \xa3100!')
|
||||
self.assertEqual(remove_entities('As low as £100!'),
|
||||
u'As low as \xa3100!')
|
||||
self.assertEqual(remove_entities('redirectTo=search&searchtext=MR0221Y&aff=buyat&affsrc=d_data&cm_mmc=buyat-_-ELECTRICAL & SEASONAL-_-MR0221Y-_-9-carat gold ½oz solid crucifix pendant'),
|
||||
u'redirectTo=search&searchtext=MR0221Y&aff=buyat&affsrc=d_data&cm_mmc=buyat-_-ELECTRICAL & SEASONAL-_-MR0221Y-_-9-carat gold \xbdoz solid crucifix pendant')
|
||||
# keep some entities
|
||||
self.assertEqual(remove_entities('<b>Low < High & Medium £ six</b>', keep=['lt', 'amp']),
|
||||
u'<b>Low < High & Medium \xa3 six</b>')
|
||||
|
||||
# illegal entities
|
||||
self.assertEqual(remove_entities('a < b &illegal; c � six', remove_illegal=False),
|
||||
u'a < b &illegal; c � six')
|
||||
self.assertEqual(remove_entities('a < b &illegal; c � six', remove_illegal=True),
|
||||
u'a < b c six')
|
||||
self.assertEqual(remove_entities('x≤y'), u'x\u2264y')
|
||||
|
||||
# check browser hack for numeric character references in the 80-9F range
|
||||
self.assertEqual(remove_entities('x™y', encoding='cp1252'), u'x\u2122y')
|
||||
|
||||
# encoding
|
||||
self.assertEqual(remove_entities('x\x99™™y', encoding='cp1252'), \
|
||||
u'x\u2122\u2122\u2122y')
|
||||
|
||||
def test_replace_tags(self):
|
||||
# make sure it always return uncode
|
||||
assert isinstance(replace_tags('no entities'), unicode)
|
||||
|
||||
self.assertEqual(replace_tags(u'This text contains <a>some tag</a>'),
|
||||
u'This text contains some tag')
|
||||
|
||||
self.assertEqual(replace_tags('This text is very im<b>port</b>ant', ' '),
|
||||
u'This text is very im port ant')
|
||||
|
||||
# multiline tags
|
||||
self.assertEqual(replace_tags('Click <a class="one"\r\n href="url">here</a>'),
|
||||
u'Click here')
|
||||
|
||||
def test_remove_comments(self):
|
||||
# make sure it always return unicode
|
||||
assert isinstance(remove_comments('without comments'), unicode)
|
||||
assert isinstance(remove_comments('<!-- with comments -->'), unicode)
|
||||
|
||||
# text without comments
|
||||
self.assertEqual(remove_comments(u'text without comments'), u'text without comments')
|
||||
|
||||
# text with comments
|
||||
self.assertEqual(remove_comments(u'<!--text with comments-->'), u'')
|
||||
self.assertEqual(remove_comments(u'Hello<!--World-->'),u'Hello')
|
||||
|
||||
def test_remove_tags(self):
|
||||
# make sure it always return unicode
|
||||
assert isinstance(remove_tags('no tags'), unicode)
|
||||
assert isinstance(remove_tags('no tags', which_ones=('p',)), unicode)
|
||||
assert isinstance(remove_tags('<p>one tag</p>'), unicode)
|
||||
assert isinstance(remove_tags('<p>one tag</p>', which_ones=('p')), unicode)
|
||||
assert isinstance(remove_tags('<a>link</a>', which_ones=('b',)), unicode)
|
||||
|
||||
# text without tags
|
||||
self.assertEqual(remove_tags(u'no tags'), u'no tags')
|
||||
self.assertEqual(remove_tags(u'no tags', which_ones=('p','b',)), u'no tags')
|
||||
|
||||
# text with tags
|
||||
self.assertEqual(remove_tags(u'<p>one p tag</p>'), u'one p tag')
|
||||
self.assertEqual(remove_tags(u'<p>one p tag</p>', which_ones=('b',)), u'<p>one p tag</p>')
|
||||
|
||||
self.assertEqual(remove_tags(u'<b>not will removed</b><i>i will removed</i>', which_ones=('i',)),
|
||||
u'<b>not will removed</b>i will removed')
|
||||
|
||||
# text with tags and attributes
|
||||
self.assertEqual(remove_tags(u'<p align="center" class="one">texty</p>'), u'texty')
|
||||
self.assertEqual(remove_tags(u'<p align="center" class="one">texty</p>', which_ones=('b',)),
|
||||
u'<p align="center" class="one">texty</p>')
|
||||
|
||||
# text with empty tags
|
||||
self.assertEqual(remove_tags(u'a<br />b<br/>c'), u'abc')
|
||||
self.assertEqual(remove_tags(u'a<br />b<br/>c', which_ones=('br',)), u'abc')
|
||||
|
||||
# test keep arg
|
||||
self.assertEqual(remove_tags(u'<p>a<br />b<br/>c</p>', keep=('br',)), u'a<br />b<br/>c')
|
||||
self.assertEqual(remove_tags(u'<p>a<br />b<br/>c</p>', keep=('p',)), u'<p>abc</p>')
|
||||
self.assertEqual(remove_tags(u'<p>a<br />b<br/>c</p>', keep=('p','br','div')), u'<p>a<br />b<br/>c</p>')
|
||||
|
||||
def test_remove_tags_with_content(self):
|
||||
# make sure it always return unicode
|
||||
assert isinstance(remove_tags_with_content('no tags'), unicode)
|
||||
assert isinstance(remove_tags_with_content('no tags', which_ones=('p',)), unicode)
|
||||
assert isinstance(remove_tags_with_content('<p>one tag</p>', which_ones=('p',)), unicode)
|
||||
assert isinstance(remove_tags_with_content('<a>link</a>', which_ones=('b',)), unicode)
|
||||
|
||||
# text without tags
|
||||
self.assertEqual(remove_tags_with_content(u'no tags'), u'no tags')
|
||||
self.assertEqual(remove_tags_with_content(u'no tags', which_ones=('p','b',)), u'no tags')
|
||||
|
||||
# text with tags
|
||||
self.assertEqual(remove_tags_with_content(u'<p>one p tag</p>'), u'<p>one p tag</p>')
|
||||
self.assertEqual(remove_tags_with_content(u'<p>one p tag</p>', which_ones=('p',)), u'')
|
||||
|
||||
self.assertEqual(remove_tags_with_content(u'<b>not will removed</b><i>i will removed</i>', which_ones=('i',)),
|
||||
u'<b>not will removed</b>')
|
||||
|
||||
# text with empty tags
|
||||
self.assertEqual(remove_tags_with_content(u'<br/>a<br />', which_ones=('br',)), u'a')
|
||||
|
||||
def test_replace_escape_chars(self):
|
||||
# make sure it always return unicode
|
||||
assert isinstance(replace_escape_chars('no ec'), unicode)
|
||||
assert isinstance(replace_escape_chars('no ec', replace_by='str'), unicode)
|
||||
assert isinstance(replace_escape_chars('no ec', which_ones=('\n','\t',)), unicode)
|
||||
|
||||
# text without escape chars
|
||||
self.assertEqual(replace_escape_chars(u'no ec'), u'no ec')
|
||||
self.assertEqual(replace_escape_chars(u'no ec', which_ones=('\n',)), u'no ec')
|
||||
|
||||
# text with escape chars
|
||||
self.assertEqual(replace_escape_chars(u'escape\n\n'), u'escape')
|
||||
self.assertEqual(replace_escape_chars(u'escape\n', which_ones=('\t',)), u'escape\n')
|
||||
self.assertEqual(replace_escape_chars(u'escape\tchars\n', which_ones=('\t')), 'escapechars\n')
|
||||
self.assertEqual(replace_escape_chars(u'escape\tchars\n', replace_by=' '), 'escape chars ')
|
||||
self.assertEqual(replace_escape_chars(u'escape\tchars\n', replace_by=u'\xa3'), u'escape\xa3chars\xa3')
|
||||
self.assertEqual(replace_escape_chars(u'escape\tchars\n', replace_by='\xc2\xa3'), u'escape\xa3chars\xa3')
|
||||
|
||||
def test_unquote_markup(self):
|
||||
sample_txt1 = u"""<node1>hi, this is sample text with entities: & ©
|
||||
<![CDATA[although this is inside a cdata! & "]]></node1>"""
|
||||
sample_txt2 = u'<node2>blah&blah<![CDATA[blahblahblah!£]]>moreblah<></node2>'
|
||||
sample_txt3 = u'something£&more<node3><![CDATA[things, stuff, and such]]>what"ever</node3><node4'
|
||||
|
||||
# make sure it always return unicode
|
||||
assert isinstance(unquote_markup(sample_txt1.encode('latin-1')), unicode)
|
||||
assert isinstance(unquote_markup(sample_txt2), unicode)
|
||||
|
||||
self.assertEqual(unquote_markup(sample_txt1), u"""<node1>hi, this is sample text with entities: & \xa9
|
||||
although this is inside a cdata! & "</node1>""")
|
||||
|
||||
self.assertEqual(unquote_markup(sample_txt2), u'<node2>blah&blahblahblahblah!£moreblah<></node2>')
|
||||
|
||||
self.assertEqual(unquote_markup(sample_txt1 + sample_txt2), u"""<node1>hi, this is sample text with entities: & \xa9
|
||||
although this is inside a cdata! & "</node1><node2>blah&blahblahblahblah!£moreblah<></node2>""")
|
||||
|
||||
self.assertEqual(unquote_markup(sample_txt3), u'something\xa3&more<node3>things, stuff, and suchwhat"ever</node3><node4')
|
||||
|
|
@ -31,108 +31,6 @@ class ResponseUtilsTest(unittest.TestCase):
|
|||
self.assertTrue(isinstance(body_or_str(u'text', unicode=False), str))
|
||||
self.assertTrue(isinstance(body_or_str(u'text', unicode=True), unicode))
|
||||
|
||||
def test_get_base_url(self):
|
||||
response = HtmlResponse(url='https://example.org', body="""\
|
||||
<html>\
|
||||
<head><title>Dummy</title><base href='http://example.org/something' /></head>\
|
||||
<body>blahablsdfsal&</body>\
|
||||
</html>""")
|
||||
self.assertEqual(get_base_url(response), 'http://example.org/something')
|
||||
|
||||
# relative url with absolute path
|
||||
response = HtmlResponse(url='https://example.org', body="""\
|
||||
<html>\
|
||||
<head><title>Dummy</title><base href='/absolutepath' /></head>\
|
||||
<body>blahablsdfsal&</body>\
|
||||
</html>""")
|
||||
self.assertEqual(get_base_url(response), 'https://example.org/absolutepath')
|
||||
|
||||
# no scheme url
|
||||
response = HtmlResponse(url='https://example.org', body="""\
|
||||
<html>\
|
||||
<head><title>Dummy</title><base href='//noscheme.com/path' /></head>\
|
||||
<body>blahablsdfsal&</body>\
|
||||
</html>""")
|
||||
self.assertEqual(get_base_url(response), 'https://noscheme.com/path')
|
||||
|
||||
def test_get_meta_refresh(self):
|
||||
body = """
|
||||
<html>
|
||||
<head><title>Dummy</title><meta http-equiv="refresh" content="5;url=http://example.org/newpage" /></head>
|
||||
<body>blahablsdfsal&</body>
|
||||
</html>"""
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (5, 'http://example.org/newpage'))
|
||||
|
||||
# refresh without url should return (None, None)
|
||||
body = """<meta http-equiv="refresh" content="5" />"""
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
|
||||
body = """<meta http-equiv="refresh" content="5;
|
||||
url=http://example.org/newpage" /></head>"""
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (5, 'http://example.org/newpage'))
|
||||
|
||||
# meta refresh in multiple lines
|
||||
body = """<html><head>
|
||||
<META
|
||||
HTTP-EQUIV="Refresh"
|
||||
CONTENT="1; URL=http://example.org/newpage">"""
|
||||
response = TextResponse(url='http://example.org', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (1, 'http://example.org/newpage'))
|
||||
|
||||
# entities in the redirect url
|
||||
body = """<meta http-equiv="refresh" content="3; url='http://www.example.com/other'">"""
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://www.example.com/other'))
|
||||
|
||||
# relative redirects
|
||||
body = """<meta http-equiv="refresh" content="3; url=other.html">"""
|
||||
response = TextResponse(url='http://example.com/page/this.html', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/page/other.html'))
|
||||
|
||||
# non-standard encodings (utf-16)
|
||||
body = """<meta http-equiv="refresh" content="3; url=http://example.com/redirect">"""
|
||||
body = body.decode('ascii').encode('utf-16')
|
||||
response = TextResponse(url='http://example.com', body=body, encoding='utf-16')
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/redirect'))
|
||||
|
||||
# non-ascii chars in the url (utf8)
|
||||
body = """<meta http-equiv="refresh" content="3; url=http://example.com/to\xc2\xa3">"""
|
||||
response = TextResponse(url='http://example.com', body=body, encoding='utf-8')
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/to%C2%A3'))
|
||||
|
||||
# non-ascii chars in the url (latin1)
|
||||
body = """<meta http-equiv="refresh" content="3; url=http://example.com/to\xa3">"""
|
||||
response = TextResponse(url='http://example.com', body=body, encoding='latin1')
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/to%C2%A3'))
|
||||
|
||||
# responses without refresh tag should return None None
|
||||
response = TextResponse(url='http://example.org')
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
response = TextResponse(url='http://example.org')
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
|
||||
# html commented meta refresh header must not directed
|
||||
body = """<!--<meta http-equiv="refresh" content="3; url=http://example.com/">-->"""
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (None, None))
|
||||
|
||||
# html comments must not interfere with uncommented meta refresh header
|
||||
body = """<!-- commented --><meta http-equiv="refresh" content="3; url=http://example.com/">-->"""
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3, 'http://example.com/'))
|
||||
|
||||
# float refresh intervals
|
||||
body = """<meta http-equiv="refresh" content=".1;URL=index.html" />"""
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (0.1, 'http://example.com/index.html'))
|
||||
|
||||
body = """<meta http-equiv="refresh" content="3.1;URL=index.html" />"""
|
||||
response = TextResponse(url='http://example.com', body=body)
|
||||
self.assertEqual(get_meta_refresh(response), (3.1, 'http://example.com/index.html'))
|
||||
|
||||
def test_response_httprepr(self):
|
||||
r1 = Response("http://www.example.com")
|
||||
self.assertEqual(response_httprepr(r1), 'HTTP/1.1 200 OK\r\n\r\n')
|
||||
|
|
|
|||
|
|
@ -1,9 +1,6 @@
|
|||
import os
|
||||
import unittest
|
||||
from scrapy.spider import BaseSpider
|
||||
from scrapy.utils.url import url_is_from_any_domain, safe_url_string, safe_download_url, \
|
||||
url_query_parameter, add_or_replace_parameter, url_query_cleaner, canonicalize_url, \
|
||||
urljoin_rfc, url_is_from_spider, file_uri_to_path, path_to_file_uri, any_to_uri
|
||||
from scrapy.utils.url import url_is_from_any_domain, url_is_from_spider, canonicalize_url
|
||||
|
||||
class UrlUtilsTest(unittest.TestCase):
|
||||
|
||||
|
|
@ -55,144 +52,6 @@ class UrlUtilsTest(unittest.TestCase):
|
|||
self.assertTrue(url_is_from_spider('http://www.example.net/some/page.html', MySpider))
|
||||
self.assertFalse(url_is_from_spider('http://www.example.us/some/page.html', MySpider))
|
||||
|
||||
def test_urljoin_rfc(self):
|
||||
self.assertEqual(urljoin_rfc('http://example.com/some/path', 'newpath/test'),
|
||||
'http://example.com/some/newpath/test')
|
||||
self.assertEqual(urljoin_rfc('http://example.com/some/path/a.jpg', '../key/other'),
|
||||
'http://example.com/some/key/other')
|
||||
u = urljoin_rfc(u'http://example.com/lolo/\xa3/lele', u'lala/\xa3')
|
||||
self.assertEqual(u, 'http://example.com/lolo/\xc2\xa3/lala/\xc2\xa3')
|
||||
assert isinstance(u, str)
|
||||
u = urljoin_rfc(u'http://example.com/lolo/\xa3/lele', 'lala/\xa3', encoding='latin-1')
|
||||
self.assertEqual(u, 'http://example.com/lolo/\xa3/lala/\xa3')
|
||||
assert isinstance(u, str)
|
||||
u = urljoin_rfc('http://example.com/lolo/\xa3/lele', 'lala/\xa3')
|
||||
self.assertEqual(u, 'http://example.com/lolo/\xa3/lala/\xa3')
|
||||
assert isinstance(u, str)
|
||||
|
||||
def test_safe_url_string(self):
|
||||
# Motoko Kusanagi (Cyborg from Ghost in the Shell)
|
||||
motoko = u'\u8349\u8599 \u7d20\u5b50'
|
||||
self.assertEqual(safe_url_string(motoko), # note the %20 for space
|
||||
'%E8%8D%89%E8%96%99%20%E7%B4%A0%E5%AD%90')
|
||||
self.assertEqual(safe_url_string(motoko),
|
||||
safe_url_string(safe_url_string(motoko)))
|
||||
self.assertEqual(safe_url_string(u'\xa9'), # copyright symbol
|
||||
'%C2%A9')
|
||||
self.assertEqual(safe_url_string(u'\xa9', 'iso-8859-1'),
|
||||
'%A9')
|
||||
self.assertEqual(safe_url_string("http://www.scrapy.org/"),
|
||||
'http://www.scrapy.org/')
|
||||
|
||||
alessi = u'/ecommerce/oggetto/Te \xf2/tea-strainer/1273'
|
||||
|
||||
self.assertEqual(safe_url_string(alessi),
|
||||
'/ecommerce/oggetto/Te%20%C3%B2/tea-strainer/1273')
|
||||
|
||||
self.assertEqual(safe_url_string("http://www.example.com/test?p(29)url(http://www.another.net/page)"),
|
||||
"http://www.example.com/test?p(29)url(http://www.another.net/page)")
|
||||
self.assertEqual(safe_url_string("http://www.example.com/Brochures_&_Paint_Cards&PageSize=200"),
|
||||
"http://www.example.com/Brochures_&_Paint_Cards&PageSize=200")
|
||||
|
||||
safeurl = safe_url_string(u"http://www.example.com/\xa3", encoding='latin-1')
|
||||
self.assert_(isinstance(safeurl, str))
|
||||
self.assertEqual(safeurl, "http://www.example.com/%A3")
|
||||
|
||||
safeurl = safe_url_string(u"http://www.example.com/\xa3", encoding='utf-8')
|
||||
self.assert_(isinstance(safeurl, str))
|
||||
self.assertEqual(safeurl, "http://www.example.com/%C2%A3")
|
||||
|
||||
def test_safe_download_url(self):
|
||||
self.assertEqual(safe_download_url('http://www.scrapy.org/../'),
|
||||
'http://www.scrapy.org/')
|
||||
self.assertEqual(safe_download_url('http://www.scrapy.org/../../images/../image'),
|
||||
'http://www.scrapy.org/image')
|
||||
self.assertEqual(safe_download_url('http://www.scrapy.org/dir/'),
|
||||
'http://www.scrapy.org/dir/')
|
||||
|
||||
def test_url_query_parameter(self):
|
||||
self.assertEqual(url_query_parameter("product.html?id=200&foo=bar", "id"),
|
||||
'200')
|
||||
self.assertEqual(url_query_parameter("product.html?id=200&foo=bar", "notthere", "mydefault"),
|
||||
'mydefault')
|
||||
self.assertEqual(url_query_parameter("product.html?id=", "id"),
|
||||
None)
|
||||
self.assertEqual(url_query_parameter("product.html?id=", "id", keep_blank_values=1),
|
||||
'')
|
||||
|
||||
def test_url_query_parameter_2(self):
|
||||
"""
|
||||
This problem was seen several times in the feeds. Sometime affiliate URLs contains
|
||||
nested encoded affiliate URL with direct URL as parameters. For example:
|
||||
aff_url1 = 'http://www.tkqlhce.com/click-2590032-10294381?url=http%3A%2F%2Fwww.argos.co.uk%2Fwebapp%2Fwcs%2Fstores%2Fservlet%2FArgosCreateReferral%3FstoreId%3D10001%26langId%3D-1%26referrer%3DCOJUN%26params%3Dadref%253DGarden+and+DIY-%3EGarden+furniture-%3EChildren%26%2339%3Bs+garden+furniture%26referredURL%3Dhttp%3A%2F%2Fwww.argos.co.uk%2Fwebapp%2Fwcs%2Fstores%2Fservlet%2FProductDisplay%253FstoreId%253D10001%2526catalogId%253D1500001501%2526productId%253D1500357023%2526langId%253D-1'
|
||||
the typical code to extract needed URL from it is:
|
||||
aff_url2 = url_query_parameter(aff_url1, 'url')
|
||||
after this aff2_url is:
|
||||
'http://www.argos.co.uk/webapp/wcs/stores/servlet/ArgosCreateReferral?storeId=10001&langId=-1&referrer=COJUN¶ms=adref%3DGarden and DIY->Garden furniture->Children's gardenfurniture&referredURL=http://www.argos.co.uk/webapp/wcs/stores/servlet/ProductDisplay%3FstoreId%3D10001%26catalogId%3D1500001501%26productId%3D1500357023%26langId%3D-1'
|
||||
the direct URL extraction is
|
||||
url = url_query_parameter(aff_url2, 'referredURL')
|
||||
but this will not work, because aff_url2 contains ' (comma sign encoded in the feed)
|
||||
and the URL extraction will fail, current workaround was made in the spider,
|
||||
just a replace for ' to %27
|
||||
"""
|
||||
return # FIXME: this test should pass but currently doesnt
|
||||
# correct case
|
||||
aff_url1 = "http://www.anrdoezrs.net/click-2590032-10294381?url=http%3A%2F%2Fwww.argos.co.uk%2Fwebapp%2Fwcs%2Fstores%2Fservlet%2FArgosCreateReferral%3FstoreId%3D10001%26langId%3D-1%26referrer%3DCOJUN%26params%3Dadref%253DGarden+and+DIY-%3EGarden+furniture-%3EGarden+table+and+chair+sets%26referredURL%3Dhttp%3A%2F%2Fwww.argos.co.uk%2Fwebapp%2Fwcs%2Fstores%2Fservlet%2FProductDisplay%253FstoreId%253D10001%2526catalogId%253D1500001501%2526productId%253D1500357199%2526langId%253D-1"
|
||||
aff_url2 = url_query_parameter(aff_url1, 'url')
|
||||
self.assertEqual(aff_url2, "http://www.argos.co.uk/webapp/wcs/stores/servlet/ArgosCreateReferral?storeId=10001&langId=-1&referrer=COJUN¶ms=adref%3DGarden and DIY->Garden furniture->Garden table and chair sets&referredURL=http://www.argos.co.uk/webapp/wcs/stores/servlet/ProductDisplay%3FstoreId%3D10001%26catalogId%3D1500001501%26productId%3D1500357199%26langId%3D-1")
|
||||
prod_url = url_query_parameter(aff_url2, 'referredURL')
|
||||
self.assertEqual(prod_url, "http://www.argos.co.uk/webapp/wcs/stores/servlet/ProductDisplay?storeId=10001&catalogId=1500001501&productId=1500357199&langId=-1")
|
||||
# weird case
|
||||
aff_url1 = "http://www.tkqlhce.com/click-2590032-10294381?url=http%3A%2F%2Fwww.argos.co.uk%2Fwebapp%2Fwcs%2Fstores%2Fservlet%2FArgosCreateReferral%3FstoreId%3D10001%26langId%3D-1%26referrer%3DCOJUN%26params%3Dadref%253DGarden+and+DIY-%3EGarden+furniture-%3EChildren%26%2339%3Bs+garden+furniture%26referredURL%3Dhttp%3A%2F%2Fwww.argos.co.uk%2Fwebapp%2Fwcs%2Fstores%2Fservlet%2FProductDisplay%253FstoreId%253D10001%2526catalogId%253D1500001501%2526productId%253D1500357023%2526langId%253D-1"
|
||||
aff_url2 = url_query_parameter(aff_url1, 'url')
|
||||
self.assertEqual(aff_url2, "http://www.argos.co.uk/webapp/wcs/stores/servlet/ArgosCreateReferral?storeId=10001&langId=-1&referrer=COJUN¶ms=adref%3DGarden and DIY->Garden furniture->Children's garden furniture&referredURL=http://www.argos.co.uk/webapp/wcs/stores/servlet/ProductDisplay%3FstoreId%3D10001%26catalogId%3D1500001501%26productId%3D1500357023%26langId%3D-1")
|
||||
prod_url = url_query_parameter(aff_url2, 'referredURL')
|
||||
# fails, prod_url is None now
|
||||
self.assertEqual(prod_url, "http://www.argos.co.uk/webapp/wcs/stores/servlet/ProductDisplay?storeId=10001&catalogId=1500001501&productId=1500357023&langId=-1")
|
||||
|
||||
def test_add_or_replace_parameter(self):
|
||||
url = 'http://domain/test'
|
||||
self.assertEqual(add_or_replace_parameter(url, 'arg', 'v'),
|
||||
'http://domain/test?arg=v')
|
||||
url = 'http://domain/test?arg1=v1&arg2=v2&arg3=v3'
|
||||
self.assertEqual(add_or_replace_parameter(url, 'arg4', 'v4'),
|
||||
'http://domain/test?arg1=v1&arg2=v2&arg3=v3&arg4=v4')
|
||||
self.assertEqual(add_or_replace_parameter(url, 'arg3', 'nv3'),
|
||||
'http://domain/test?arg1=v1&arg2=v2&arg3=nv3')
|
||||
url = 'http://domain/test?arg1=v1'
|
||||
self.assertEqual(add_or_replace_parameter(url, 'arg2', 'v2', sep=';'),
|
||||
'http://domain/test?arg1=v1;arg2=v2')
|
||||
self.assertEqual(add_or_replace_parameter("http://domain/moreInfo.asp?prodID=", 'prodID', '20'),
|
||||
'http://domain/moreInfo.asp?prodID=20')
|
||||
url = 'http://rmc-offers.co.uk/productlist.asp?BCat=2%2C60&CatID=60'
|
||||
self.assertEqual(add_or_replace_parameter(url, 'BCat', 'newvalue', url_is_quoted=True),
|
||||
'http://rmc-offers.co.uk/productlist.asp?BCat=newvalue&CatID=60')
|
||||
url = 'http://rmc-offers.co.uk/productlist.asp?BCat=2,60&CatID=60'
|
||||
self.assertEqual(add_or_replace_parameter(url, 'BCat', 'newvalue'),
|
||||
'http://rmc-offers.co.uk/productlist.asp?BCat=newvalue&CatID=60')
|
||||
|
||||
def test_url_query_cleaner(self):
|
||||
self.assertEqual('product.html?id=200',
|
||||
url_query_cleaner("product.html?id=200&foo=bar&name=wired", ['id']))
|
||||
self.assertEqual('product.html?id=200',
|
||||
url_query_cleaner("product.html?&id=200&&foo=bar&name=wired", ['id']))
|
||||
self.assertEqual('product.html',
|
||||
url_query_cleaner("product.html?foo=bar&name=wired", ['id']))
|
||||
self.assertEqual('product.html?id=200&name=wired',
|
||||
url_query_cleaner("product.html?id=200&foo=bar&name=wired", ['id', 'name']))
|
||||
self.assertEqual('product.html?id',
|
||||
url_query_cleaner("product.html?id&other=3&novalue=", ['id']))
|
||||
self.assertEqual('product.html?d=1&d=2&d=3',
|
||||
url_query_cleaner("product.html?d=1&e=b&d=2&d=3&other=other", ['d'], unique=False))
|
||||
self.assertEqual('product.html?id=200&foo=bar',
|
||||
url_query_cleaner("product.html?id=200&foo=bar&name=wired#id20", ['id', 'foo']))
|
||||
self.assertEqual('product.html?foo=bar&name=wired',
|
||||
url_query_cleaner("product.html?id=200&foo=bar&name=wired", ['id'], remove=True))
|
||||
self.assertEqual('product.html?name=wired',
|
||||
url_query_cleaner("product.html?id=2&foo=bar&name=wired", ['id', 'foo'], remove=True))
|
||||
self.assertEqual('product.html?foo=bar&name=wired',
|
||||
url_query_cleaner("product.html?id=2&foo=bar&name=wired", ['id', 'footo'], remove=True))
|
||||
|
||||
def test_canonicalize_url(self):
|
||||
# simplest case
|
||||
self.assertEqual(canonicalize_url("http://www.example.com"),
|
||||
|
|
@ -283,50 +142,6 @@ class UrlUtilsTest(unittest.TestCase):
|
|||
self.assertEqual(canonicalize_url("http://www.EXAMPLE.com"),
|
||||
"http://www.example.com")
|
||||
|
||||
def test_path_to_file_uri(self):
|
||||
if os.name == 'nt':
|
||||
self.assertEqual(path_to_file_uri("C:\\windows\clock.avi"),
|
||||
"file:///C:/windows/clock.avi")
|
||||
else:
|
||||
self.assertEqual(path_to_file_uri("/some/path.txt"),
|
||||
"file:///some/path.txt")
|
||||
|
||||
fn = "test.txt"
|
||||
x = path_to_file_uri(fn)
|
||||
self.assert_(x.startswith('file:///'))
|
||||
self.assertEqual(file_uri_to_path(x).lower(), os.path.abspath(fn).lower())
|
||||
|
||||
def test_file_uri_to_path(self):
|
||||
if os.name == 'nt':
|
||||
self.assertEqual(file_uri_to_path("file:///C:/windows/clock.avi"),
|
||||
"C:\\windows\clock.avi")
|
||||
uri = "file:///C:/windows/clock.avi"
|
||||
uri2 = path_to_file_uri(file_uri_to_path(uri))
|
||||
self.assertEqual(uri, uri2)
|
||||
else:
|
||||
self.assertEqual(file_uri_to_path("file:///path/to/test.txt"),
|
||||
"/path/to/test.txt")
|
||||
self.assertEqual(file_uri_to_path("/path/to/test.txt"),
|
||||
"/path/to/test.txt")
|
||||
uri = "file:///path/to/test.txt"
|
||||
uri2 = path_to_file_uri(file_uri_to_path(uri))
|
||||
self.assertEqual(uri, uri2)
|
||||
|
||||
self.assertEqual(file_uri_to_path("test.txt"),
|
||||
"test.txt")
|
||||
|
||||
def test_any_to_uri(self):
|
||||
if os.name == 'nt':
|
||||
self.assertEqual(any_to_uri("C:\\windows\clock.avi"),
|
||||
"file:///C:/windows/clock.avi")
|
||||
else:
|
||||
self.assertEqual(any_to_uri("/some/path.txt"),
|
||||
"file:///some/path.txt")
|
||||
self.assertEqual(any_to_uri("file:///some/path.txt"),
|
||||
"file:///some/path.txt")
|
||||
self.assertEqual(any_to_uri("http://www.example.com/some/path.txt"),
|
||||
"http://www.example.com/some/path.txt")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
|
|
@ -1,61 +1,7 @@
|
|||
from base64 import urlsafe_b64encode
|
||||
"""
|
||||
Transitional module for moving to the w3lib library.
|
||||
|
||||
def headers_raw_to_dict(headers_raw):
|
||||
"""
|
||||
Convert raw headers (single multi-line string)
|
||||
to the dictionary.
|
||||
For new code, always import from w3lib.http instead of this module
|
||||
"""
|
||||
|
||||
For example:
|
||||
>>> headers_raw_to_dict("Content-type: text/html\\n\\rAccept: gzip\\n\\n")
|
||||
{'Content-type': ['text/html'], 'Accept': ['gzip']}
|
||||
|
||||
Incorrect input:
|
||||
>>> headers_raw_to_dict("Content-typt gzip\\n\\n")
|
||||
{}
|
||||
|
||||
Argument is None:
|
||||
>>> headers_raw_to_dict(None)
|
||||
"""
|
||||
if headers_raw is None:
|
||||
return None
|
||||
return dict([
|
||||
(header_item[0].strip(), [header_item[1].strip()])
|
||||
for header_item
|
||||
in [
|
||||
header.split(':', 1)
|
||||
for header
|
||||
in headers_raw.splitlines()]
|
||||
if len(header_item) == 2])
|
||||
|
||||
|
||||
def headers_dict_to_raw(headers_dict):
|
||||
"""
|
||||
Returns a raw HTTP headers representation of headers
|
||||
|
||||
For example:
|
||||
>>> headers_dict_to_raw({'Content-type': 'text/html', 'Accept': 'gzip'})
|
||||
'Content-type: text/html\\r\\nAccept: gzip'
|
||||
>>> from twisted.python.util import InsensitiveDict
|
||||
>>> td = InsensitiveDict({'Content-type': ['text/html'], 'Accept': ['gzip']})
|
||||
>>> headers_dict_to_raw(td)
|
||||
'Content-type: text/html\\r\\nAccept: gzip'
|
||||
|
||||
Argument is None:
|
||||
>>> headers_dict_to_raw(None)
|
||||
|
||||
"""
|
||||
if headers_dict is None:
|
||||
return None
|
||||
raw_lines = []
|
||||
for key, value in headers_dict.items():
|
||||
if isinstance(value, (str, unicode)):
|
||||
raw_lines.append("%s: %s" % (key, value))
|
||||
elif isinstance(value, (list, tuple)):
|
||||
for v in value:
|
||||
raw_lines.append("%s: %s" % (key, v))
|
||||
return '\r\n'.join(raw_lines)
|
||||
|
||||
|
||||
def basic_auth_header(username, password):
|
||||
"""Return `Authorization` header for HTTP Basic Access Authentication (RFC 2617)"""
|
||||
return 'Basic ' + urlsafe_b64encode("%s:%s" % (username, password))
|
||||
from w3lib.http import *
|
||||
|
|
|
|||
|
|
@ -1,165 +1,7 @@
|
|||
"""
|
||||
Functions for dealing with markup text
|
||||
Transitional module for moving to the w3lib library.
|
||||
|
||||
For new code, always import from w3lib.html instead of this module
|
||||
"""
|
||||
|
||||
import re
|
||||
from htmlentitydefs import name2codepoint
|
||||
|
||||
from scrapy.utils.python import str_to_unicode
|
||||
|
||||
_ent_re = re.compile(r'&(#?(x?))([^&;\s]+);')
|
||||
_tag_re = re.compile(r'<[a-zA-Z\/!].*?>', re.DOTALL)
|
||||
|
||||
def remove_entities(text, keep=(), remove_illegal=True, encoding='utf-8'):
|
||||
"""Remove entities from the given text.
|
||||
|
||||
'text' can be a unicode string or a regular string encoded in the given
|
||||
`encoding` (which defaults to 'utf-8').
|
||||
|
||||
If 'keep' is passed (with a list of entity names) those entities will
|
||||
be kept (they won't be removed).
|
||||
|
||||
It supports both numeric (&#nnnn; and &#hhhh;) and named ( >)
|
||||
entities.
|
||||
|
||||
If remove_illegal is True, entities that can't be converted are removed.
|
||||
If remove_illegal is False, entities that can't be converted are kept "as
|
||||
is". For more information see the tests.
|
||||
|
||||
Always returns a unicode string (with the entities removed).
|
||||
"""
|
||||
|
||||
def convert_entity(m):
|
||||
entity_body = m.group(3)
|
||||
if m.group(1):
|
||||
try:
|
||||
if m.group(2):
|
||||
number = int(entity_body, 16)
|
||||
else:
|
||||
number = int(entity_body, 10)
|
||||
# Numeric character references in the 80-9F range are typically
|
||||
# interpreted by browsers as representing the characters mapped
|
||||
# to bytes 80-9F in the Windows-1252 encoding. For more info
|
||||
# see: http://en.wikipedia.org/wiki/Character_encodings_in_HTML
|
||||
if 0x80 <= number <= 0x9f:
|
||||
return chr(number).decode('cp1252')
|
||||
except ValueError:
|
||||
number = None
|
||||
else:
|
||||
if entity_body in keep:
|
||||
return m.group(0)
|
||||
else:
|
||||
number = name2codepoint.get(entity_body)
|
||||
if number is not None:
|
||||
try:
|
||||
return unichr(number)
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
return u'' if remove_illegal else m.group(0)
|
||||
|
||||
return _ent_re.sub(convert_entity, str_to_unicode(text, encoding))
|
||||
|
||||
def has_entities(text, encoding=None):
|
||||
return bool(_ent_re.search(str_to_unicode(text, encoding)))
|
||||
|
||||
def replace_tags(text, token='', encoding=None):
|
||||
"""Replace all markup tags found in the given text by the given token. By
|
||||
default token is a null string so it just remove all tags.
|
||||
|
||||
'text' can be a unicode string or a regular string encoded as 'utf-8'
|
||||
|
||||
Always returns a unicode string.
|
||||
"""
|
||||
return _tag_re.sub(token, str_to_unicode(text, encoding))
|
||||
|
||||
|
||||
def remove_comments(text, encoding=None):
|
||||
""" Remove HTML Comments. """
|
||||
return re.sub('<!--.*?-->', u'', str_to_unicode(text, encoding), re.DOTALL)
|
||||
|
||||
def remove_tags(text, which_ones=(), keep=(), encoding=None):
|
||||
""" Remove HTML Tags only.
|
||||
|
||||
which_ones and keep are both tuples, there are four cases:
|
||||
|
||||
which_ones, keep (1 - not empty, 0 - empty)
|
||||
1, 0 - remove all tags in which_ones
|
||||
0, 1 - remove all tags except the ones in keep
|
||||
0, 0 - remove all tags
|
||||
1, 1 - not allowd
|
||||
"""
|
||||
|
||||
assert not (which_ones and keep), 'which_ones and keep can not be given at the same time'
|
||||
|
||||
def will_remove(tag):
|
||||
if which_ones:
|
||||
return tag in which_ones
|
||||
else:
|
||||
return tag not in keep
|
||||
|
||||
def remove_tag(m):
|
||||
tag = m.group(1)
|
||||
return u'' if will_remove(tag) else m.group(0)
|
||||
|
||||
regex = '</?([^ >/]+).*?>'
|
||||
retags = re.compile(regex, re.DOTALL | re.IGNORECASE)
|
||||
|
||||
return retags.sub(remove_tag, str_to_unicode(text, encoding))
|
||||
|
||||
def remove_tags_with_content(text, which_ones=(), encoding=None):
|
||||
""" Remove tags and its content.
|
||||
|
||||
which_ones -- is a tuple of which tags with its content we want to remove.
|
||||
if is empty do nothing.
|
||||
"""
|
||||
text = str_to_unicode(text, encoding)
|
||||
if which_ones:
|
||||
tags = '|'.join([r'<%s.*?</%s>|<%s\s*/>' % (tag, tag, tag) for tag in which_ones])
|
||||
retags = re.compile(tags, re.DOTALL | re.IGNORECASE)
|
||||
text = retags.sub(u'', text)
|
||||
return text
|
||||
|
||||
|
||||
def replace_escape_chars(text, which_ones=('\n', '\t', '\r'), replace_by=u'', \
|
||||
encoding=None):
|
||||
""" Remove escape chars. Default : \\n, \\t, \\r
|
||||
|
||||
which_ones -- is a tuple of which escape chars we want to remove.
|
||||
By default removes \n, \t, \r.
|
||||
|
||||
replace_by -- text to replace the escape chars for.
|
||||
It defaults to '', so the escape chars are removed.
|
||||
"""
|
||||
for ec in which_ones:
|
||||
text = text.replace(ec, str_to_unicode(replace_by, encoding))
|
||||
return str_to_unicode(text, encoding)
|
||||
|
||||
def unquote_markup(text, keep=(), remove_illegal=True, encoding=None):
|
||||
"""
|
||||
This function receives markup as a text (always a unicode string or a utf-8 encoded string) and does the following:
|
||||
- removes entities (except the ones in 'keep') from any part of it that it's not inside a CDATA
|
||||
- searches for CDATAs and extracts their text (if any) without modifying it.
|
||||
- removes the found CDATAs
|
||||
"""
|
||||
_cdata_re = re.compile(r'((?P<cdata_s><!\[CDATA\[)(?P<cdata_d>.*?)(?P<cdata_e>\]\]>))', re.DOTALL)
|
||||
|
||||
def _get_fragments(txt, pattern):
|
||||
offset = 0
|
||||
for match in pattern.finditer(txt):
|
||||
match_s, match_e = match.span(1)
|
||||
yield txt[offset:match_s]
|
||||
yield match
|
||||
offset = match_e
|
||||
yield txt[offset:]
|
||||
|
||||
text = str_to_unicode(text, encoding)
|
||||
ret_text = u''
|
||||
for fragment in _get_fragments(text, _cdata_re):
|
||||
if isinstance(fragment, basestring):
|
||||
# it's not a CDATA (so we try to remove its entities)
|
||||
ret_text += remove_entities(fragment, keep=keep, remove_illegal=remove_illegal)
|
||||
else:
|
||||
# it's a CDATA (so we just extract its content)
|
||||
ret_text += fragment.group('cdata_d')
|
||||
return ret_text
|
||||
from w3lib.html import *
|
||||
|
|
|
|||
|
|
@ -1,34 +1,7 @@
|
|||
from cStringIO import StringIO
|
||||
"""
|
||||
Transitional module for moving to the w3lib library.
|
||||
|
||||
def encode_multipart(data):
|
||||
"""Encode the given data to be used in a multipart HTTP POST. Data is a
|
||||
where keys are the field name, and values are either strings or tuples
|
||||
(filename, content) for file uploads.
|
||||
For new code, always import from w3lib.form instead of this module
|
||||
"""
|
||||
|
||||
This code is based on distutils.command.upload
|
||||
"""
|
||||
|
||||
# Build up the MIME payload for the POST data
|
||||
boundary = '--------------GHSKFJDLGDS7543FJKLFHRE75642756743254'
|
||||
sep_boundary = '\r\n--' + boundary
|
||||
end_boundary = sep_boundary + '--'
|
||||
body = StringIO()
|
||||
for key, value in data.items():
|
||||
# handle multiple entries for the same name
|
||||
if type(value) != type([]):
|
||||
value = [value]
|
||||
for value in value:
|
||||
if type(value) is tuple:
|
||||
fn = '; filename="%s"' % value[0]
|
||||
value = value[1]
|
||||
else:
|
||||
fn = ""
|
||||
|
||||
body.write(sep_boundary)
|
||||
body.write('\r\nContent-Disposition: form-data; name="%s"' % key)
|
||||
body.write(fn)
|
||||
body.write("\r\n\r\n")
|
||||
body.write(value)
|
||||
body.write(end_boundary)
|
||||
body.write("\r\n")
|
||||
return body.getvalue(), boundary
|
||||
from w3lib.form import *
|
||||
|
|
|
|||
|
|
@ -4,16 +4,14 @@ scrapy.http.Response objects
|
|||
"""
|
||||
|
||||
import os
|
||||
import re
|
||||
import weakref
|
||||
import webbrowser
|
||||
import tempfile
|
||||
|
||||
from twisted.web import http
|
||||
from twisted.web.http import RESPONSES
|
||||
from w3lib import html
|
||||
|
||||
from scrapy.utils.markup import remove_entities, remove_comments
|
||||
from scrapy.utils.url import safe_url_string, urljoin_rfc
|
||||
from scrapy.xlib.BeautifulSoup import BeautifulSoup
|
||||
from scrapy.http import Response, HtmlResponse
|
||||
|
||||
|
|
@ -27,37 +25,22 @@ def body_or_str(obj, unicode=True):
|
|||
else:
|
||||
return obj if unicode else obj.encode('utf-8')
|
||||
|
||||
BASEURL_RE = re.compile(ur'<base\s+href\s*=\s*[\"\']\s*([^\"\'\s]+)\s*[\"\']', re.I)
|
||||
_baseurl_cache = weakref.WeakKeyDictionary()
|
||||
def get_base_url(response):
|
||||
""" Return the base url of the given response used to resolve relative links. """
|
||||
"""Return the base url of the given response, joined with the response url"""
|
||||
if response not in _baseurl_cache:
|
||||
match = BASEURL_RE.search(response.body_as_unicode()[0:4096])
|
||||
_baseurl_cache[response] = urljoin_rfc(response.url, match.group(1)) if match else response.url
|
||||
text = response.body_as_unicode()[0:4096]
|
||||
_baseurl_cache[response] = html.get_base_url(text, response.url, \
|
||||
response.encoding)
|
||||
return _baseurl_cache[response]
|
||||
|
||||
META_REFRESH_RE = re.compile(ur'<meta[^>]*http-equiv[^>]*refresh[^>]*content\s*=\s*(?P<quote>["\'])(?P<int>(\d*\.)?\d+)\s*;\s*url=(?P<url>.*?)(?P=quote)', \
|
||||
re.DOTALL | re.IGNORECASE)
|
||||
_metaref_cache = weakref.WeakKeyDictionary()
|
||||
def get_meta_refresh(response):
|
||||
"""Parse the http-equiv parameter of the HTML meta element from the given
|
||||
response and return a tuple (interval, url) where interval is an integer
|
||||
containing the delay in seconds (or zero if not present) and url is a
|
||||
string with the absolute url to redirect.
|
||||
|
||||
If no meta redirect is found, (None, None) is returned.
|
||||
"""
|
||||
"""Parse the http-equiv refrsh parameter from the given response"""
|
||||
if response not in _metaref_cache:
|
||||
body_chunk = remove_comments(remove_entities(response.body_as_unicode()[0:4096]))
|
||||
match = META_REFRESH_RE.search(body_chunk)
|
||||
if match:
|
||||
interval = float(match.group('int'))
|
||||
url = safe_url_string(match.group('url').strip(' "\''))
|
||||
url = urljoin_rfc(response.url, url)
|
||||
_metaref_cache[response] = (interval, url)
|
||||
else:
|
||||
_metaref_cache[response] = (None, None)
|
||||
#_metaref_cache[response] = match.groups() if match else (None, None)
|
||||
text = response.body_as_unicode()[0:4096]
|
||||
_metaref_cache[response] = html.get_meta_refresh(text, response.url, \
|
||||
response.encoding)
|
||||
return _metaref_cache[response]
|
||||
|
||||
_beautifulsoup_cache = weakref.WeakKeyDictionary()
|
||||
|
|
|
|||
|
|
@ -1,6 +1,9 @@
|
|||
"""
|
||||
This module contains general purpose URL functions not found in the standard
|
||||
library.
|
||||
|
||||
Some of the functions that used to be imported from this module have been moved
|
||||
to the w3lib.url module. Always import those from there instead.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
|
@ -10,6 +13,7 @@ import urllib
|
|||
import posixpath
|
||||
import cgi
|
||||
|
||||
from w3lib.url import *
|
||||
from scrapy.utils.python import unicode_to_str
|
||||
|
||||
def url_is_from_any_domain(url, domains):
|
||||
|
|
@ -26,109 +30,6 @@ def url_is_from_spider(url, spider):
|
|||
return url_is_from_any_domain(url, [spider.name] + \
|
||||
getattr(spider, 'allowed_domains', []))
|
||||
|
||||
def urljoin_rfc(base, ref, encoding='utf-8'):
|
||||
"""Same as urlparse.urljoin but supports unicode values in base and ref
|
||||
parameters (in which case they will be converted to str using the given
|
||||
encoding).
|
||||
|
||||
Always returns a str.
|
||||
"""
|
||||
return urlparse.urljoin(unicode_to_str(base, encoding), \
|
||||
unicode_to_str(ref, encoding))
|
||||
|
||||
_reserved = ';/?:@&=+$|,#' # RFC 3986 (Generic Syntax)
|
||||
_unreserved_marks = "-_.!~*'()" # RFC 3986 sec 2.3
|
||||
_safe_chars = urllib.always_safe + '%' + _reserved + _unreserved_marks
|
||||
|
||||
def safe_url_string(url, encoding='utf8'):
|
||||
"""Convert the given url into a legal URL by escaping unsafe characters
|
||||
according to RFC-3986.
|
||||
|
||||
If a unicode url is given, it is first converted to str using the given
|
||||
encoding (which defaults to 'utf-8'). When passing a encoding, you should
|
||||
use the encoding of the original page (the page from which the url was
|
||||
extracted from).
|
||||
|
||||
Calling this function on an already "safe" url will return the url
|
||||
unmodified.
|
||||
|
||||
Always returns a str.
|
||||
"""
|
||||
s = unicode_to_str(url, encoding)
|
||||
return urllib.quote(s, _safe_chars)
|
||||
|
||||
|
||||
_parent_dirs = re.compile(r'/?(\.\./)+')
|
||||
|
||||
def safe_download_url(url):
|
||||
""" Make a url for download. This will call safe_url_string
|
||||
and then strip the fragment, if one exists. The path will
|
||||
be normalised.
|
||||
|
||||
If the path is outside the document root, it will be changed
|
||||
to be within the document root.
|
||||
"""
|
||||
safe_url = safe_url_string(url)
|
||||
scheme, netloc, path, query, _ = urlparse.urlsplit(safe_url)
|
||||
if path:
|
||||
path = _parent_dirs.sub('', posixpath.normpath(path))
|
||||
if url.endswith('/') and not path.endswith('/'):
|
||||
path += '/'
|
||||
else:
|
||||
path = '/'
|
||||
return urlparse.urlunsplit((scheme, netloc, path, query, ''))
|
||||
|
||||
def is_url(text):
|
||||
return text.partition("://")[0] in ('file', 'http', 'https')
|
||||
|
||||
def url_query_parameter(url, parameter, default=None, keep_blank_values=0):
|
||||
"""Return the value of a url parameter, given the url and parameter name"""
|
||||
queryparams = cgi.parse_qs(urlparse.urlsplit(str(url))[3], \
|
||||
keep_blank_values=keep_blank_values)
|
||||
return queryparams.get(parameter, [default])[0]
|
||||
|
||||
def url_query_cleaner(url, parameterlist=(), sep='&', kvsep='=', remove=False, unique=True):
|
||||
"""Clean url arguments leaving only those passed in the parameterlist keeping order
|
||||
|
||||
If remove is True, leave only those not in parameterlist.
|
||||
If unique is False, do not remove duplicated keys
|
||||
"""
|
||||
url = urlparse.urldefrag(url)[0]
|
||||
base, _, query = url.partition('?')
|
||||
seen = set()
|
||||
querylist = []
|
||||
for ksv in query.split(sep):
|
||||
k, _, _ = ksv.partition(kvsep)
|
||||
if unique and k in seen:
|
||||
continue
|
||||
elif remove and k in parameterlist:
|
||||
continue
|
||||
elif not remove and k not in parameterlist:
|
||||
continue
|
||||
else:
|
||||
querylist.append(ksv)
|
||||
seen.add(k)
|
||||
return '?'.join([base, sep.join(querylist)]) if querylist else base
|
||||
|
||||
def add_or_replace_parameter(url, name, new_value, sep='&', url_is_quoted=False):
|
||||
"""Add or remove a parameter to a given url"""
|
||||
def has_querystring(url):
|
||||
_, _, _, query, _ = urlparse.urlsplit(url)
|
||||
return bool(query)
|
||||
|
||||
parameter = url_query_parameter(url, name, keep_blank_values=1)
|
||||
if url_is_quoted:
|
||||
parameter = urllib.quote(parameter)
|
||||
if parameter is None:
|
||||
if has_querystring(url):
|
||||
next_url = url + sep + name + '=' + new_value
|
||||
else:
|
||||
next_url = url + '?' + name + '=' + new_value
|
||||
else:
|
||||
next_url = url.replace(name+'='+parameter,
|
||||
name+'='+new_value)
|
||||
return next_url
|
||||
|
||||
def canonicalize_url(url, keep_blank_values=True, keep_fragments=False, \
|
||||
encoding=None):
|
||||
"""Canonicalize the given url by applying the following procedures:
|
||||
|
|
@ -155,27 +56,3 @@ def canonicalize_url(url, keep_blank_values=True, keep_fragments=False, \
|
|||
path = safe_url_string(urllib.unquote(path))
|
||||
fragment = '' if not keep_fragments else fragment
|
||||
return urlparse.urlunparse((scheme, netloc.lower(), path, params, query, fragment))
|
||||
|
||||
def path_to_file_uri(path):
|
||||
"""Convert local filesystem path to legal File URIs as described in:
|
||||
http://en.wikipedia.org/wiki/File_URI_scheme
|
||||
"""
|
||||
x = urllib.pathname2url(os.path.abspath(path))
|
||||
if os.name == 'nt':
|
||||
x = x.replace('|', ':') # http://bugs.python.org/issue5861
|
||||
return 'file:///%s' % x.lstrip('/')
|
||||
|
||||
def file_uri_to_path(uri):
|
||||
"""Convert File URI to local filesystem path according to:
|
||||
http://en.wikipedia.org/wiki/File_URI_scheme
|
||||
"""
|
||||
return urllib.url2pathname(urlparse.urlparse(uri).path)
|
||||
|
||||
def any_to_uri(uri_or_path):
|
||||
"""If given a path name, return its File URI, otherwise return it
|
||||
unmodified
|
||||
"""
|
||||
if os.path.splitdrive(uri_or_path)[0]:
|
||||
return path_to_file_uri(uri_or_path)
|
||||
u = urlparse.urlparse(uri_or_path)
|
||||
return uri_or_path if u.scheme else path_to_file_uri(uri_or_path)
|
||||
|
|
|
|||
2
setup.py
2
setup.py
|
|
@ -120,7 +120,7 @@ setup_args = {
|
|||
|
||||
try:
|
||||
from setuptools import setup
|
||||
setup_args['install_requires'] = ['Twisted>=2.5', 'lxml']
|
||||
setup_args['install_requires'] = ['Twisted>=2.5', 'lxml', 'w3lib']
|
||||
if sys.version_info < (2, 6):
|
||||
setup_args['install_requires'] += ['simplejson']
|
||||
except ImportError:
|
||||
|
|
|
|||
Loading…
Reference in New Issue