From c476681c0608fce9eec523d446daf142be3a95e0 Mon Sep 17 00:00:00 2001 From: Pablo Hoffman Date: Tue, 21 Feb 2012 21:31:19 -0200 Subject: [PATCH] ported to code to use w3lib.encoding (work in progress, many tests failing yet) --- scrapy/http/response/dammit.py | 269 ------------------ scrapy/http/response/html.py | 21 +- scrapy/http/response/text.py | 34 +-- scrapy/http/response/xml.py | 15 +- scrapy/settings/default_settings.py | 34 --- ...st_downloadermiddleware_httpcompression.py | 2 +- scrapy/tests/test_http_response.py | 2 +- scrapy/tests/test_utils_encoding.py | 25 -- scrapy/utils/encoding.py | 15 +- 9 files changed, 18 insertions(+), 399 deletions(-) delete mode 100644 scrapy/http/response/dammit.py delete mode 100644 scrapy/tests/test_utils_encoding.py diff --git a/scrapy/http/response/dammit.py b/scrapy/http/response/dammit.py deleted file mode 100644 index b9254904d..000000000 --- a/scrapy/http/response/dammit.py +++ /dev/null @@ -1,269 +0,0 @@ -""" -This module contains a fork of the UnicodeDammit class from BeautifulSoup, that -expliclty disabled any usage of chardet library. - -The UnicodeDammit class is used as a last resource for detecting the encoding -of a response. -""" - -import re -import codecs - -chardet = None # we don't want to use chardet since it's very slow, - -class UnicodeDammit: - """A class for detecting the encoding of a *ML document and - converting it to a Unicode string. If the source encoding is - windows-1252, can replace MS smart quotes with their HTML or XML - equivalents.""" - - # This dictionary maps commonly seen values for "charset" in HTML - # meta tags to the corresponding Python codec names. It only covers - # values that aren't in Python's aliases and can't be determined - # by the heuristics in find_codec. - CHARSET_ALIASES = { "macintosh" : "mac-roman", - "x-sjis" : "shift-jis" } - - def __init__(self, markup, overrideEncodings=[], - smartQuotesTo='xml', isHTML=False): - self.declaredHTMLEncoding = None - self.markup, documentEncoding, sniffedEncoding = \ - self._detectEncoding(markup, isHTML) - self.smartQuotesTo = smartQuotesTo - self.triedEncodings = [] - if markup == '' or isinstance(markup, unicode): - self.originalEncoding = None - self.unicode = unicode(markup) - return - - u = None - for proposedEncoding in overrideEncodings: - u = self._convertFrom(proposedEncoding) - if u: break - if not u: - for proposedEncoding in (documentEncoding, sniffedEncoding): - u = self._convertFrom(proposedEncoding) - if u: break - - # If no luck and we have auto-detection library, try that: - if not u and chardet and not isinstance(self.markup, unicode): - u = self._convertFrom(chardet.detect(self.markup)['encoding']) - - # As a last resort, try utf-8 and windows-1252: - if not u: - for proposed_encoding in ("utf-8", "windows-1252"): - u = self._convertFrom(proposed_encoding) - if u: break - - self.unicode = u - if not u: self.originalEncoding = None - - def _subMSChar(self, orig): - """Changes a MS smart quote character to an XML or HTML - entity.""" - sub = self.MS_CHARS.get(orig) - if isinstance(sub, tuple): - if self.smartQuotesTo == 'xml': - sub = '&#x%s;' % sub[1] - else: - sub = '&%s;' % sub[0] - return sub - - def _convertFrom(self, proposed): - proposed = self.find_codec(proposed) - if not proposed or proposed in self.triedEncodings: - return None - self.triedEncodings.append(proposed) - markup = self.markup - - # Convert smart quotes to HTML if coming from an encoding - # that might have them. - if self.smartQuotesTo and proposed.lower() in("windows-1252", - "iso-8859-1", - "iso-8859-2"): - markup = re.compile("([\x80-\x9f])").sub \ - (lambda(x): self._subMSChar(x.group(1)), - markup) - - try: - # print "Trying to convert document to %s" % proposed - u = self._toUnicode(markup, proposed) - self.markup = u - self.originalEncoding = proposed - except Exception, e: - # print "That didn't work!" - # print e - return None - #print "Correct encoding: %s" % proposed - return self.markup - - def _toUnicode(self, data, encoding): - '''Given a string and its encoding, decodes the string into Unicode. - %encoding is a string recognized by encodings.aliases''' - - # strip Byte Order Mark (if present) - if (len(data) >= 4) and (data[:2] == '\xfe\xff') \ - and (data[2:4] != '\x00\x00'): - encoding = 'utf-16be' - data = data[2:] - elif (len(data) >= 4) and (data[:2] == '\xff\xfe') \ - and (data[2:4] != '\x00\x00'): - encoding = 'utf-16le' - data = data[2:] - elif data[:3] == '\xef\xbb\xbf': - encoding = 'utf-8' - data = data[3:] - elif data[:4] == '\x00\x00\xfe\xff': - encoding = 'utf-32be' - data = data[4:] - elif data[:4] == '\xff\xfe\x00\x00': - encoding = 'utf-32le' - data = data[4:] - newdata = unicode(data, encoding) - return newdata - - def _detectEncoding(self, xml_data, isHTML=False): - """Given a document, tries to detect its XML encoding.""" - xml_encoding = sniffed_xml_encoding = None - try: - if xml_data[:4] == '\x4c\x6f\xa7\x94': - # EBCDIC - xml_data = self._ebcdic_to_ascii(xml_data) - elif xml_data[:4] == '\x00\x3c\x00\x3f': - # UTF-16BE - sniffed_xml_encoding = 'utf-16be' - xml_data = unicode(xml_data, 'utf-16be').encode('utf-8') - elif (len(xml_data) >= 4) and (xml_data[:2] == '\xfe\xff') \ - and (xml_data[2:4] != '\x00\x00'): - # UTF-16BE with BOM - sniffed_xml_encoding = 'utf-16be' - xml_data = unicode(xml_data[2:], 'utf-16be').encode('utf-8') - elif xml_data[:4] == '\x3c\x00\x3f\x00': - # UTF-16LE - sniffed_xml_encoding = 'utf-16le' - xml_data = unicode(xml_data, 'utf-16le').encode('utf-8') - elif (len(xml_data) >= 4) and (xml_data[:2] == '\xff\xfe') and \ - (xml_data[2:4] != '\x00\x00'): - # UTF-16LE with BOM - sniffed_xml_encoding = 'utf-16le' - xml_data = unicode(xml_data[2:], 'utf-16le').encode('utf-8') - elif xml_data[:4] == '\x00\x00\x00\x3c': - # UTF-32BE - sniffed_xml_encoding = 'utf-32be' - xml_data = unicode(xml_data, 'utf-32be').encode('utf-8') - elif xml_data[:4] == '\x3c\x00\x00\x00': - # UTF-32LE - sniffed_xml_encoding = 'utf-32le' - xml_data = unicode(xml_data, 'utf-32le').encode('utf-8') - elif xml_data[:4] == '\x00\x00\xfe\xff': - # UTF-32BE with BOM - sniffed_xml_encoding = 'utf-32be' - xml_data = unicode(xml_data[4:], 'utf-32be').encode('utf-8') - elif xml_data[:4] == '\xff\xfe\x00\x00': - # UTF-32LE with BOM - sniffed_xml_encoding = 'utf-32le' - xml_data = unicode(xml_data[4:], 'utf-32le').encode('utf-8') - elif xml_data[:3] == '\xef\xbb\xbf': - # UTF-8 with BOM - sniffed_xml_encoding = 'utf-8' - xml_data = unicode(xml_data[3:], 'utf-8').encode('utf-8') - else: - sniffed_xml_encoding = 'ascii' - pass - except: - xml_encoding_match = None - xml_encoding_match = re.compile( - '^<\?.*encoding=[\'"](.*?)[\'"].*\?>').match(xml_data) - if not xml_encoding_match and isHTML: - regexp = re.compile('<\s*meta[^>]+charset=([^>]*?)[;\'">]', re.I) - xml_encoding_match = regexp.search(xml_data) - if xml_encoding_match is not None: - xml_encoding = xml_encoding_match.groups()[0].lower() - if isHTML: - self.declaredHTMLEncoding = xml_encoding - if sniffed_xml_encoding and \ - (xml_encoding in ('iso-10646-ucs-2', 'ucs-2', 'csunicode', - 'iso-10646-ucs-4', 'ucs-4', 'csucs4', - 'utf-16', 'utf-32', 'utf_16', 'utf_32', - 'utf16', 'u16')): - xml_encoding = sniffed_xml_encoding - return xml_data, xml_encoding, sniffed_xml_encoding - - - def find_codec(self, charset): - return self._codec(self.CHARSET_ALIASES.get(charset, charset)) \ - or (charset and self._codec(charset.replace("-", ""))) \ - or (charset and self._codec(charset.replace("-", "_"))) \ - or charset - - def _codec(self, charset): - if not charset: return charset - codec = None - try: - codecs.lookup(charset) - codec = charset - except (LookupError, ValueError): - pass - return codec - - EBCDIC_TO_ASCII_MAP = None - def _ebcdic_to_ascii(self, s): - c = self.__class__ - if not c.EBCDIC_TO_ASCII_MAP: - emap = (0,1,2,3,156,9,134,127,151,141,142,11,12,13,14,15, - 16,17,18,19,157,133,8,135,24,25,146,143,28,29,30,31, - 128,129,130,131,132,10,23,27,136,137,138,139,140,5,6,7, - 144,145,22,147,148,149,150,4,152,153,154,155,20,21,158,26, - 32,160,161,162,163,164,165,166,167,168,91,46,60,40,43,33, - 38,169,170,171,172,173,174,175,176,177,93,36,42,41,59,94, - 45,47,178,179,180,181,182,183,184,185,124,44,37,95,62,63, - 186,187,188,189,190,191,192,193,194,96,58,35,64,39,61,34, - 195,97,98,99,100,101,102,103,104,105,196,197,198,199,200, - 201,202,106,107,108,109,110,111,112,113,114,203,204,205, - 206,207,208,209,126,115,116,117,118,119,120,121,122,210, - 211,212,213,214,215,216,217,218,219,220,221,222,223,224, - 225,226,227,228,229,230,231,123,65,66,67,68,69,70,71,72, - 73,232,233,234,235,236,237,125,74,75,76,77,78,79,80,81, - 82,238,239,240,241,242,243,92,159,83,84,85,86,87,88,89, - 90,244,245,246,247,248,249,48,49,50,51,52,53,54,55,56,57, - 250,251,252,253,254,255) - import string - c.EBCDIC_TO_ASCII_MAP = string.maketrans( \ - ''.join(map(chr, range(256))), ''.join(map(chr, emap))) - return s.translate(c.EBCDIC_TO_ASCII_MAP) - - MS_CHARS = { '\x80' : ('euro', '20AC'), - '\x81' : ' ', - '\x82' : ('sbquo', '201A'), - '\x83' : ('fnof', '192'), - '\x84' : ('bdquo', '201E'), - '\x85' : ('hellip', '2026'), - '\x86' : ('dagger', '2020'), - '\x87' : ('Dagger', '2021'), - '\x88' : ('circ', '2C6'), - '\x89' : ('permil', '2030'), - '\x8A' : ('Scaron', '160'), - '\x8B' : ('lsaquo', '2039'), - '\x8C' : ('OElig', '152'), - '\x8D' : '?', - '\x8E' : ('#x17D', '17D'), - '\x8F' : '?', - '\x90' : '?', - '\x91' : ('lsquo', '2018'), - '\x92' : ('rsquo', '2019'), - '\x93' : ('ldquo', '201C'), - '\x94' : ('rdquo', '201D'), - '\x95' : ('bull', '2022'), - '\x96' : ('ndash', '2013'), - '\x97' : ('mdash', '2014'), - '\x98' : ('tilde', '2DC'), - '\x99' : ('trade', '2122'), - '\x9a' : ('scaron', '161'), - '\x9b' : ('rsaquo', '203A'), - '\x9c' : ('oelig', '153'), - '\x9d' : '?', - '\x9e' : ('#x17E', '17E'), - '\x9f' : ('Yuml', ''),} - -####################################################################### - diff --git a/scrapy/http/response/html.py b/scrapy/http/response/html.py index f4236f426..bd3559fbb 100644 --- a/scrapy/http/response/html.py +++ b/scrapy/http/response/html.py @@ -5,26 +5,7 @@ discovering through HTML encoding declarations to the TextResponse class. See documentation in docs/topics/request-response.rst """ -import re - from scrapy.http.response.text import TextResponse -from scrapy.utils.python import memoizemethod_noargs class HtmlResponse(TextResponse): - - _template = r'''%s\s*=\s*["']?\s*%s\s*["']?''' - - _httpequiv_re = _template % ('http-equiv', 'Content-Type') - _content_re = _template % ('content', r'(?P[^;]+);\s*charset=(?P[\w-]+)') - _content2_re = _template % ('charset', r'(?P[\w-]+)') - - METATAG_RE = re.compile(r'[\w-]+)') - XMLDECL_RE = re.compile(r'<\?xml\s.*?%s' % _encoding_re, re.I) - - @memoizemethod_noargs - def _body_declared_encoding(self): - chunk = self.body[:5000] - match = self.XMLDECL_RE.search(chunk) - return match.group('charset') if match else None - + pass diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index 278ce8570..ba8ac704d 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -98,40 +98,6 @@ except KeyError: else: EDITOR = 'vi' -ENCODING_ALIASES = {} - -ENCODING_ALIASES_BASE = { - # gb2312 is superseded by gb18030 - 'gb2312': 'gb18030', - 'chinese': 'gb18030', - 'csiso58gb231280': 'gb18030', - 'euc- cn': 'gb18030', - 'euccn': 'gb18030', - 'eucgb2312-cn': 'gb18030', - 'gb2312-1980': 'gb18030', - 'gb2312-80': 'gb18030', - 'iso- ir-58': 'gb18030', - # gbk is superseded by gb18030 - 'gbk': 'gb18030', - '936': 'gb18030', - 'cp936': 'gb18030', - 'ms936': 'gb18030', - # latin_1 is a subset of cp1252 - 'latin_1': 'cp1252', - 'iso-8859-1': 'cp1252', - 'iso8859-1': 'cp1252', - '8859': 'cp1252', - 'cp819': 'cp1252', - 'latin': 'cp1252', - 'latin1': 'cp1252', - 'l1': 'cp1252', - # others - 'zh-cn': 'gb18030', - 'win-1251': 'cp1251', - 'macintosh' : 'mac_roman', - 'x-sjis': 'shift_jis', -} - EXTENSIONS = {} EXTENSIONS_BASE = { diff --git a/scrapy/tests/test_downloadermiddleware_httpcompression.py b/scrapy/tests/test_downloadermiddleware_httpcompression.py index 98255db16..4385fbae1 100644 --- a/scrapy/tests/test_downloadermiddleware_httpcompression.py +++ b/scrapy/tests/test_downloadermiddleware_httpcompression.py @@ -9,7 +9,7 @@ from scrapy.spider import BaseSpider from scrapy.http import Response, Request, HtmlResponse from scrapy.contrib.downloadermiddleware.httpcompression import HttpCompressionMiddleware from scrapy.tests import tests_datadir -from scrapy.utils.encoding import resolve_encoding +from w3lib.encoding import resolve_encoding SAMPLEDIR = join(tests_datadir, 'compressed') diff --git a/scrapy/tests/test_http_response.py b/scrapy/tests/test_http_response.py index 46314c1b1..fdf923996 100644 --- a/scrapy/tests/test_http_response.py +++ b/scrapy/tests/test_http_response.py @@ -1,7 +1,7 @@ import unittest +from w3lib.encoding import resolve_encoding from scrapy.http import Request, Response, TextResponse, HtmlResponse, XmlResponse, Headers -from scrapy.utils.encoding import resolve_encoding class BaseResponseTest(unittest.TestCase): diff --git a/scrapy/tests/test_utils_encoding.py b/scrapy/tests/test_utils_encoding.py deleted file mode 100644 index b6800cd75..000000000 --- a/scrapy/tests/test_utils_encoding.py +++ /dev/null @@ -1,25 +0,0 @@ -import unittest - -from scrapy.utils.encoding import encoding_exists, resolve_encoding - -class UtilsEncodingTestCase(unittest.TestCase): - - _ENCODING_ALIASES = { - 'foo': 'cp1252', - 'bar': 'none', - } - - def test_resolve_encoding(self): - self.assertEqual(resolve_encoding('latin1', self._ENCODING_ALIASES), - 'latin1') - self.assertEqual(resolve_encoding('foo', self._ENCODING_ALIASES), - 'cp1252') - - def test_encoding_exists(self): - assert encoding_exists('latin1', self._ENCODING_ALIASES) - assert encoding_exists('foo', self._ENCODING_ALIASES) - assert not encoding_exists('bar', self._ENCODING_ALIASES) - assert not encoding_exists('none', self._ENCODING_ALIASES) - -if __name__ == "__main__": - unittest.main() diff --git a/scrapy/utils/encoding.py b/scrapy/utils/encoding.py index c7d645041..7e33943cd 100644 --- a/scrapy/utils/encoding.py +++ b/scrapy/utils/encoding.py @@ -1,20 +1,11 @@ import codecs -from scrapy.conf import settings +from w3lib.encoding import resolve_encoding -_ENCODING_ALIASES = dict(settings['ENCODING_ALIASES_BASE']) -_ENCODING_ALIASES.update(settings['ENCODING_ALIASES']) - -def encoding_exists(encoding, _aliases=_ENCODING_ALIASES): +def encoding_exists(encoding): """Returns ``True`` if encoding is valid, otherwise returns ``False``""" try: - codecs.lookup(resolve_encoding(encoding, _aliases)) + codecs.lookup(resolve_encoding(encoding)) except LookupError: return False return True - -def resolve_encoding(alias, _aliases=_ENCODING_ALIASES): - """Return the encoding the given alias maps to, or the alias as passed if - no mapping is found. - """ - return _aliases.get(alias.lower(), alias)