mirror of https://github.com/scrapy/scrapy.git
151 lines
8.0 KiB
Python
151 lines
8.0 KiB
Python
import unittest
|
||
from scrapy.responsetypes import responsetypes
|
||
|
||
from scrapy.http import Response, TextResponse, XmlResponse, HtmlResponse, Headers
|
||
|
||
|
||
class ResponseTypesTest(unittest.TestCase):
|
||
|
||
def test_from_filename(self):
|
||
mappings = [
|
||
('data.bin', Response),
|
||
('file.txt', TextResponse),
|
||
('file.xml.gz', Response),
|
||
('file.xml', XmlResponse),
|
||
('file.html', HtmlResponse),
|
||
('file.unknownext', Response),
|
||
]
|
||
for source, cls in mappings:
|
||
retcls = responsetypes.from_filename(source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_from_content_disposition(self):
|
||
mappings = [
|
||
(b'attachment; filename="data.xml"', XmlResponse),
|
||
(b'attachment; filename=data.xml', XmlResponse),
|
||
('attachment;filename=data£.tar.gz'.encode('utf-8'), Response),
|
||
('attachment;filename=dataµ.tar.gz'.encode('latin-1'), Response),
|
||
('attachment;filename=data高.doc'.encode('gbk'), Response),
|
||
('attachment;filename=دورهdata.html'.encode('cp720'), HtmlResponse),
|
||
('attachment;filename=日本語版Wikipedia.xml'.encode('iso2022_jp'), XmlResponse),
|
||
|
||
]
|
||
for source, cls in mappings:
|
||
retcls = responsetypes.from_content_disposition(source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_from_content_type(self):
|
||
mappings = [
|
||
('text/html; charset=UTF-8', HtmlResponse),
|
||
('text/xml; charset=UTF-8', XmlResponse),
|
||
('application/xhtml+xml; charset=UTF-8', HtmlResponse),
|
||
('application/vnd.wap.xhtml+xml; charset=utf-8', HtmlResponse),
|
||
('application/xml; charset=UTF-8', XmlResponse),
|
||
('application/octet-stream', Response),
|
||
('application/x-json; encoding=UTF8;charset=UTF-8', TextResponse),
|
||
('application/json-amazonui-streaming;charset=UTF-8', TextResponse),
|
||
]
|
||
for source, cls in mappings:
|
||
retcls = responsetypes.from_content_type(source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_from_body(self):
|
||
mappings = [
|
||
(b'\x03\x02\xdf\xdd\x23', Response),
|
||
(b'Some plain text\ndata with tabs\t and null bytes\0', TextResponse),
|
||
(b'<html><head><title>Hello</title></head>', HtmlResponse),
|
||
# https://codersblock.com/blog/the-smallest-valid-html5-page/
|
||
(b'<!DOCTYPE html>\n<title>.</title>', HtmlResponse),
|
||
(b'<?xml version="1.0" encoding="utf-8"', XmlResponse),
|
||
]
|
||
for source, cls in mappings:
|
||
retcls = responsetypes.from_body(source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_from_headers(self):
|
||
mappings = [
|
||
({'Content-Type': ['text/html; charset=utf-8']}, HtmlResponse),
|
||
({'Content-Type': ['text/html; charset=utf-8'], 'Content-Encoding': ['gzip']}, Response),
|
||
({'Content-Type': ['application/octet-stream'],
|
||
'Content-Disposition': ['attachment; filename=data.txt']}, TextResponse),
|
||
]
|
||
for source, cls in mappings:
|
||
source = Headers(source)
|
||
retcls = responsetypes.from_headers(source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_from_args_pre_xtractmime(self):
|
||
"""Each of the following test cases remains unaffected after
|
||
using xtractmime for MIME sniffing"""
|
||
mappings = [
|
||
({'url': 'http://www.example.com/data.csv'}, TextResponse),
|
||
({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}),
|
||
'url': 'http://www.example.com/item/'}, HtmlResponse),
|
||
({'body': b'\x01\x02', 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}),
|
||
'url': 'http://www.example.com/page/'}, Response),
|
||
({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse),
|
||
({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})},
|
||
Response),
|
||
({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/',
|
||
'headers': Headers({'Content-Type': ['text/plain']})}, TextResponse),
|
||
({'url': 'http://www.example.com/item/file.html'}, HtmlResponse),
|
||
({'body': b'<html><head><title>Hello</title></head>'}, HtmlResponse),
|
||
({'body': b'<?xml version="1.0" encoding="utf-8"'}, XmlResponse),
|
||
({'body': b'Some plain text data\1\2 with tabs and\n null bytes\0'}, Response),
|
||
({'body': b'\x01\x02', 'headers': Headers({'Content-Type': ['application/pdf']})}, Response),
|
||
({'headers': Headers({'Content-Type': ['application/x-json']})}, TextResponse),
|
||
({'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse),
|
||
({'headers': Headers({'Content-Type': ['application/json-amazonui-streaming']})}, TextResponse),
|
||
({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}),
|
||
'url': 'http://www.example.com/page/'}, Response),
|
||
({'headers': Headers({'Content-Type': ['application/pdf']})}, Response),
|
||
({'url': 'http://www.example.com/page/file.html',
|
||
'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse),
|
||
({'url': 'http://www.example.com/item/file.xml',
|
||
'headers': Headers({'Content-Type': 'application/octet-stream'})}, XmlResponse),
|
||
({'url': 'http://www.example.com/item/file.html',
|
||
'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse),
|
||
({'url': 'http://www.example.com/item/file.xml',
|
||
'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"'],
|
||
'Content-Type': 'application/octet-stream'})}, XmlResponse),
|
||
({'url': 'http://www.example.com/item/file.pdf'}, Response),
|
||
({'filename': 'file.pdf'}, Response),
|
||
]
|
||
for source, cls in mappings:
|
||
retcls = responsetypes.from_args(**source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_from_args_post_xtractmime(self):
|
||
"""Each of the following test cases got affected after
|
||
using xtractmime for MIME sniffing"""
|
||
mappings = [
|
||
({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/',
|
||
'headers': Headers({'Content-Type': ['text/plain']})}, Response),
|
||
({'filename': '/tmp/temp^'}, TextResponse),
|
||
({'body': b'%PDF-1.4'}, Response),
|
||
({'headers': Headers({'Content-Type': ['application/ecmascript']})}, TextResponse),
|
||
({'headers': Headers({'Content-Type': ['application/ld+json']})}, TextResponse),
|
||
({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})},
|
||
HtmlResponse),
|
||
({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']})},
|
||
TextResponse),
|
||
({'body': b'Non HTML', 'headers': Headers({'Content-Encoding': ['zip'],
|
||
'Content-Type': ['text/html']})}, HtmlResponse),
|
||
({'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'})},
|
||
Response),
|
||
({'body': b'\x0c\x1b'}, TextResponse),
|
||
({'body': b'this is not <html>'}, TextResponse),
|
||
({'body': b'this is not <?xml'}, TextResponse),
|
||
]
|
||
for source, cls in mappings:
|
||
retcls = responsetypes.from_args(**source)
|
||
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
|
||
|
||
def test_custom_mime_types_loaded(self):
|
||
# check that mime.types files shipped with scrapy are loaded
|
||
self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test')
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|