import os import unittest from urllib.parse import urlparse import pytest from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse from scrapy.http.headers import Headers from scrapy.utils.python import to_bytes from scrapy.utils.response import ( get_meta_refresh, get_response_class, open_in_browser, response_httprepr, response_status_message, ) __doctests__ = ['scrapy.utils.response'] # Scenarios that work the same with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args PRE_XTRACTMIME_SCENARIOS = ( ( { 'url': 'http://www.example.com/data.csv', }, TextResponse, ), ( { 'url': 'http://www.example.com/item/', 'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), }, HtmlResponse, ), ( { 'url': 'http://www.example.com/page/', 'headers': Headers( { 'Content-Disposition': [ 'attachment; filename="data.xml.gz"', ] } ), 'body': b'\x01\x02', }, Response, ), ( {'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse, ), ( { 'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'}), }, Response, ), ( { 'body': b'\x00\x01\xff', 'url': '://www.example.com/item/', 'headers': Headers({'Content-Type': ['text/plain']}), }, TextResponse, ), ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), ({'body': b'Hello'}, HtmlResponse), ({'body': b'\n.'}, HtmlResponse), ( { 'body': b'\x01\x02', 'headers': Headers({'Content-Type': ['application/pdf']}), }, Response, ), ( {'headers': Headers({'Content-Type': ['application/x-json']})}, TextResponse, ), ( {'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse, ), ( { 'headers': Headers( {'Content-Type': ['application/json-amazonui-streaming']} ) }, TextResponse, ), ( { 'headers': Headers( {'Content-Disposition': ['attachment; filename="data.xml.gz"']} ), 'url': 'http://www.example.com/page/', }, Response, ), ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), ( { 'url': 'http://www.example.com/page/file.html', 'headers': Headers({'Content-Type': 'application/octet-stream'}), }, HtmlResponse, ), ( { 'url': 'http://www.example.com/item/file.xml', 'headers': Headers({'Content-Type': 'application/octet-stream'}), }, XmlResponse, ), ( { 'url': 'http://www.example.com/item/file.html', 'headers': Headers({'Content-Type': 'application/octet-stream'}), }, HtmlResponse, ), ( { 'url': 'http://www.example.com/item/file.xml', 'headers': Headers( { 'Content-Disposition': [ 'attachment; filename="data.xml.gz"' ], 'Content-Type': 'application/octet-stream', } ), }, XmlResponse, ), ({'url': 'http://www.example.com/item/file.pdf'}, Response), ({'filename': 'file.pdf'}, Response), ( { 'body': b'\n.', 'url': 'http://www.example.com', }, HtmlResponse, ), ) # Scenarios that work differently with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args POST_XTRACTMIME_SCENARIOS = ( ( { 'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/', 'headers': Headers({'Content-Type': ['text/plain']}), }, Response, ), ({'filename': '/tmp/temp^'}, TextResponse), ({'body': b'%PDF-1.4'}, Response), ( {'headers': Headers({'Content-Type': ['application/ecmascript']})}, TextResponse, ), ( {'headers': Headers({'Content-Type': ['application/ld+json']})}, TextResponse, ), ( { 'headers': Headers( {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} ) }, HtmlResponse, ), ( { 'headers': Headers( {'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']} ) }, TextResponse, ), ( { 'body': b'Non HTML', 'headers': Headers( {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} ), }, HtmlResponse, ), ( { 'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'}), }, Response, ), ({'body': b'\x0c\x1b'}, TextResponse), ({'body': b'this is not '}, TextResponse), ({'body': b'this is not test page test body " def browser_open(burl): path = urlparse(burl).path if not os.path.exists(path): path = burl.replace('file://', '') with open(path, "rb") as f: bbody = f.read() self.assertIn(b'', bbody) return True response = HtmlResponse(url, body=body) assert open_in_browser(response, _openfunc=browser_open), "Browser not called" resp = Response(url, body=body) self.assertRaises(TypeError, open_in_browser, resp, debug=True) def test_get_meta_refresh(self): r1 = HtmlResponse("http://www.example.com", body=b""" Dummy blahablsdfsal& """) r2 = HtmlResponse("http://www.example.com", body=b""" Dummy blahablsdfsal& """) r3 = HtmlResponse("http://www.example.com", body=b"""