scrapy/tests/test_responsetypes.py

151 lines
8.0 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

import unittest
from scrapy.responsetypes import responsetypes
from scrapy.http import Response, TextResponse, XmlResponse, HtmlResponse, Headers
class ResponseTypesTest(unittest.TestCase):
def test_from_filename(self):
mappings = [
('data.bin', Response),
('file.txt', TextResponse),
('file.xml.gz', Response),
('file.xml', XmlResponse),
('file.html', HtmlResponse),
('file.unknownext', Response),
]
for source, cls in mappings:
retcls = responsetypes.from_filename(source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_from_content_disposition(self):
mappings = [
(b'attachment; filename="data.xml"', XmlResponse),
(b'attachment; filename=data.xml', XmlResponse),
('attachment;filename=data£.tar.gz'.encode('utf-8'), Response),
('attachment;filename=dataµ.tar.gz'.encode('latin-1'), Response),
('attachment;filename=data高.doc'.encode('gbk'), Response),
('attachment;filename=دورهdata.html'.encode('cp720'), HtmlResponse),
('attachment;filename=日本語版Wikipedia.xml'.encode('iso2022_jp'), XmlResponse),
]
for source, cls in mappings:
retcls = responsetypes.from_content_disposition(source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_from_content_type(self):
mappings = [
('text/html; charset=UTF-8', HtmlResponse),
('text/xml; charset=UTF-8', XmlResponse),
('application/xhtml+xml; charset=UTF-8', HtmlResponse),
('application/vnd.wap.xhtml+xml; charset=utf-8', HtmlResponse),
('application/xml; charset=UTF-8', XmlResponse),
('application/octet-stream', Response),
('application/x-json; encoding=UTF8;charset=UTF-8', TextResponse),
('application/json-amazonui-streaming;charset=UTF-8', TextResponse),
]
for source, cls in mappings:
retcls = responsetypes.from_content_type(source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_from_body(self):
mappings = [
(b'\x03\x02\xdf\xdd\x23', Response),
(b'Some plain text\ndata with tabs\t and null bytes\0', TextResponse),
(b'<html><head><title>Hello</title></head>', HtmlResponse),
# https://codersblock.com/blog/the-smallest-valid-html5-page/
(b'<!DOCTYPE html>\n<title>.</title>', HtmlResponse),
(b'<?xml version="1.0" encoding="utf-8"', XmlResponse),
]
for source, cls in mappings:
retcls = responsetypes.from_body(source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_from_headers(self):
mappings = [
({'Content-Type': ['text/html; charset=utf-8']}, HtmlResponse),
({'Content-Type': ['text/html; charset=utf-8'], 'Content-Encoding': ['gzip']}, Response),
({'Content-Type': ['application/octet-stream'],
'Content-Disposition': ['attachment; filename=data.txt']}, TextResponse),
]
for source, cls in mappings:
source = Headers(source)
retcls = responsetypes.from_headers(source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_from_args_pre_xtractmime(self):
"""Each of the following test cases remains unaffected after
using xtractmime for MIME sniffing"""
mappings = [
({'url': 'http://www.example.com/data.csv'}, TextResponse),
({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}),
'url': 'http://www.example.com/item/'}, HtmlResponse),
({'body': b'\x01\x02', 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}),
'url': 'http://www.example.com/page/'}, Response),
({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse),
({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})},
Response),
({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/',
'headers': Headers({'Content-Type': ['text/plain']})}, TextResponse),
({'url': 'http://www.example.com/item/file.html'}, HtmlResponse),
({'body': b'<html><head><title>Hello</title></head>'}, HtmlResponse),
({'body': b'<?xml version="1.0" encoding="utf-8"'}, XmlResponse),
({'body': b'Some plain text data\1\2 with tabs and\n null bytes\0'}, Response),
({'body': b'\x01\x02', 'headers': Headers({'Content-Type': ['application/pdf']})}, Response),
({'headers': Headers({'Content-Type': ['application/x-json']})}, TextResponse),
({'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse),
({'headers': Headers({'Content-Type': ['application/json-amazonui-streaming']})}, TextResponse),
({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}),
'url': 'http://www.example.com/page/'}, Response),
({'headers': Headers({'Content-Type': ['application/pdf']})}, Response),
({'url': 'http://www.example.com/page/file.html',
'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse),
({'url': 'http://www.example.com/item/file.xml',
'headers': Headers({'Content-Type': 'application/octet-stream'})}, XmlResponse),
({'url': 'http://www.example.com/item/file.html',
'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse),
({'url': 'http://www.example.com/item/file.xml',
'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"'],
'Content-Type': 'application/octet-stream'})}, XmlResponse),
({'url': 'http://www.example.com/item/file.pdf'}, Response),
({'filename': 'file.pdf'}, Response),
]
for source, cls in mappings:
retcls = responsetypes.from_args(**source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_from_args_post_xtractmime(self):
"""Each of the following test cases got affected after
using xtractmime for MIME sniffing"""
mappings = [
({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/',
'headers': Headers({'Content-Type': ['text/plain']})}, Response),
({'filename': '/tmp/temp^'}, TextResponse),
({'body': b'%PDF-1.4'}, Response),
({'headers': Headers({'Content-Type': ['application/ecmascript']})}, TextResponse),
({'headers': Headers({'Content-Type': ['application/ld+json']})}, TextResponse),
({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})},
HtmlResponse),
({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']})},
TextResponse),
({'body': b'Non HTML', 'headers': Headers({'Content-Encoding': ['zip'],
'Content-Type': ['text/html']})}, HtmlResponse),
({'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'})},
Response),
({'body': b'\x0c\x1b'}, TextResponse),
({'body': b'this is not <html>'}, TextResponse),
({'body': b'this is not <?xml'}, TextResponse),
]
for source, cls in mappings:
retcls = responsetypes.from_args(**source)
assert retcls is cls, f"{source} ==> {retcls} != {cls}"
def test_custom_mime_types_loaded(self):
# check that mime.types files shipped with scrapy are loaded
self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test')
if __name__ == "__main__":
unittest.main()