Merge pull request #1121 from nyov/nyov/py3-urllib

[MRG+1] more python3 compatibility changes for urllib
This commit is contained in:
Julia Medina 2015-04-02 22:33:43 -03:00
commit fcf97001a5
5 changed files with 28 additions and 23 deletions

View File

@ -5,10 +5,11 @@ import glob
import tempfile
import shutil
import time
import urllib2
import netrc
import json
from six.moves.urllib import request
from six.moves.urllib.parse import urlparse, urljoin
from six.moves.urllib.error import HTTPError, URLError
from subprocess import Popen, PIPE, check_call
from w3lib.form import encode_multipart
@ -69,7 +70,7 @@ class Command(ScrapyCommand):
except ImportError:
raise UsageError("setuptools not installed")
urllib2.install_opener(urllib2.build_opener(HTTPRedirectHandler))
request.install_opener(request.build_opener(HTTPRedirectHandler))
if opts.list_targets:
for name, target in _get_targets().items():
@ -78,9 +79,9 @@ class Command(ScrapyCommand):
if opts.list_projects:
target = _get_target(opts.list_projects)
req = urllib2.Request(_url(target, 'listprojects.json'))
req = request.Request(_url(target, 'listprojects.json'))
_add_auth_header(req, target)
f = urllib2.urlopen(req)
f = request.urlopen(req)
projects = json.loads(f.read())['projects']
print(os.linesep.join(projects))
return
@ -188,7 +189,7 @@ def _upload_egg(target, eggpath, project, version):
'Content-Type': 'multipart/form-data; boundary=%s' % boundary,
'Content-Length': str(len(body)),
}
req = urllib2.Request(url, body, headers)
req = request.Request(url, body, headers)
_add_auth_header(req, target)
_log('Deploying to project "%s" in %s' % (project, url))
return _http_post(req)
@ -207,14 +208,14 @@ def _add_auth_header(request, target):
def _http_post(request):
try:
f = urllib2.urlopen(request)
f = request.urlopen(request)
_log("Server response (%s):" % f.code)
print(f.read())
return True
except urllib2.HTTPError as e:
except HTTPError as e:
_log("Deploy failed (%s):" % e.code)
print(e.read())
except urllib2.URLError as e:
except URLError as e:
_log("Deploy failed: %s" % e)
def _build_egg():
@ -237,12 +238,12 @@ def _create_default_setup_py(**kwargs):
f.write(_SETUP_PY_TEMPLATE % kwargs)
class HTTPRedirectHandler(urllib2.HTTPRedirectHandler):
class HTTPRedirectHandler(request.HTTPRedirectHandler):
def redirect_request(self, req, fp, code, msg, headers, newurl):
newurl = newurl.replace(' ', '%20')
if code in (301, 307):
return urllib2.Request(newurl,
return request.Request(newurl,
data=req.get_data(),
headers=req.headers,
origin_req_host=req.get_origin_req_host(),
@ -250,9 +251,9 @@ class HTTPRedirectHandler(urllib2.HTTPRedirectHandler):
elif code in (302, 303):
newheaders = dict((k, v) for k, v in req.headers.items()
if k.lower() not in ("content-length", "content-type"))
return urllib2.Request(newurl,
return request.Request(newurl,
headers=newheaders,
origin_req_host=req.get_origin_req_host(),
unverifiable=True)
else:
raise urllib2.HTTPError(req.get_full_url(), code, msg, headers, fp)
raise HTTPError(req.get_full_url(), code, msg, headers, fp)

View File

@ -1,6 +1,10 @@
import base64
from urllib import getproxies, unquote, proxy_bypass
from urllib2 import _parse_proxy
from six.moves.urllib.request import getproxies, proxy_bypass
from six.moves.urllib.parse import unquote
try:
from urllib2 import _parse_proxy
except ImportError:
from urllib.request import _parse_proxy
from six.moves.urllib.parse import urlunparse
from scrapy.utils.httpobj import urlparse_cached

View File

@ -5,8 +5,7 @@ This module implements the FormRequest class which is a more convenient class
See documentation in docs/topics/request-response.rst
"""
import urllib
from six.moves.urllib.parse import urljoin
from six.moves.urllib.parse import urljoin, urlencode
import lxml.html
import six
from scrapy.http.request import Request
@ -52,7 +51,7 @@ def _urlencode(seq, enc):
values = [(unicode_to_str(k, enc), unicode_to_str(v, enc))
for k, vs in seq
for v in (vs if hasattr(vs, '__iter__') else [vs])]
return urllib.urlencode(values, doseq=1)
return urlencode(values, doseq=1)
def _get_form(response, formname, formnumber, formxpath):

View File

@ -7,8 +7,8 @@ to the w3lib.url module. Always import those from there instead.
"""
import posixpath
from six.moves.urllib.parse import (ParseResult, urlunparse, urldefrag,
urlparse, parse_qsl)
import urllib
urlparse, parse_qsl, urlencode,
unquote)
# scrapy.utils.url was moved to w3lib.url and import * ensures this move doesn't break old code
from w3lib.url import *
@ -56,7 +56,7 @@ def canonicalize_url(url, keep_blank_values=True, keep_fragments=False,
scheme, netloc, path, params, query, fragment = parse_url(url)
keyvals = parse_qsl(query, keep_blank_values)
keyvals.sort()
query = urllib.urlencode(keyvals)
query = urlencode(keyvals)
path = safe_url_string(_unquotepath(path)) or '/'
fragment = '' if not keep_fragments else fragment
return urlunparse((scheme, netloc.lower(), path, params, query, fragment))
@ -65,7 +65,7 @@ def canonicalize_url(url, keep_blank_values=True, keep_fragments=False,
def _unquotepath(path):
for reserved in ('2f', '2F', '3f', '3F'):
path = path.replace('%' + reserved, '%25' + reserved.upper())
return urllib.unquote(path)
return unquote(path)
def parse_url(url, encoding=None):

View File

@ -1,6 +1,7 @@
from __future__ import print_function
import sys, time, random, urllib, os, json
import sys, time, random, os, json
import six
from six.moves.urllib.parse import urlencode
from subprocess import Popen, PIPE
from twisted.web.server import Site, NOT_DONE_YET
from twisted.web.resource import Resource
@ -73,7 +74,7 @@ class Follow(LeafResource):
args = request.args.copy()
for nl in nlist:
args["n"] = [str(nl)]
argstr = urllib.urlencode(args, doseq=True)
argstr = urlencode(args, doseq=True)
s += "<a href='/follow?%s'>follow %d</a><br>" % (argstr, nl)
s += """</body>"""
request.write(s)