diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 632787e06..fd7310c80 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -9,6 +9,7 @@ from urllib.parse import urlparse from warnings import warn from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime +from xtractmime._utils import contains_binary from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response @@ -127,7 +128,11 @@ class ResponseTypes: filename = url if filename: - return (self.mimetypes.guess_type(filename)[0].encode(),) + mimetype, encoding = self.mimetypes.guess_type(filename) + if encoding: + return (f"application/{encoding}".encode(),) + else: + return (mimetype.encode(),) return None @@ -137,7 +142,16 @@ class ResponseTypes: if not body: body = b'' - body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") + contains_binary_bytes = False + + for index in range(len(body)): + if body[index:index + 1] != b"\x00" and contains_binary(body[index:index + 1]): + contains_binary_bytes = True + break + + if not contains_binary_bytes: + body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") + cls = Response http_origin = not url or urlparse(url).scheme in ("http", "https") content_types = self._guess_content_type(headers=headers, url=url, filename=filename) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 69ee1341e..9a07dc2f8 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -76,9 +76,9 @@ class ResponseTypesTest(unittest.TestCase): mappings = [ ({'url': 'http://www.example.com/data.csv'}, TextResponse), ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), - 'url': 'http://www.example.com/item/'}, HtmlResponse), # Failing with xtractmime, returning TextResponse expected HtmlResponse + 'url': 'http://www.example.com/item/'}, HtmlResponse), ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), - 'url': 'http://www.example.com/page/'}, Response), # Failing with xtractmime, returning TextResponse expected Response + 'url': 'http://www.example.com/page/'}, Response), ] for source, cls in mappings: retcls = responsetypes.from_args(**source) @@ -86,19 +86,19 @@ class ResponseTypesTest(unittest.TestCase): def test_from_args_post_xtractmime(self): mappings = [ - ({'body': b'Some plain text data with tabs and null bytes'}, TextResponse), - ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, + ({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse), + ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), # different behaviour with http and non-http urls - ({'body': b'\x00\xfe\xff', 'url': 'http://www.example.com/item/', - 'headers': Headers({'Content-Type': b'text/plain'})}, Response), - ({'body': b'\x00\xfe\xff', 'url': '://www.example.com/item/', + ({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/', + 'headers': Headers({'Content-Type': b'text/plain'})}, Response), + ({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/', 'headers': Headers({'Content-Type': b'text/plain'})}, TextResponse), - ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), # Failing with xtractmime, return TextResponse expected HtmlResponse - ({'body': b'Some plain text\ndata with tabs\t and null bytes\0'}, Response), # earlier expected to be binary response, refer "test_from_body()" + ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), + ({'body': b'Some plain text data\1 with tabs and\n null bytes\0'}, Response), ({'body': b'Hello'}, HtmlResponse), ({'body': b'