import os
import unittest
from urllib.parse import urlparse
import pytest
from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse
from scrapy.http.headers import Headers
from scrapy.utils.python import to_bytes
from scrapy.utils.response import (
get_meta_refresh,
get_response_class,
open_in_browser,
response_httprepr,
response_status_message,
)
__doctests__ = ['scrapy.utils.response']
# Scenarios that work the same with the previously-used, deprecated
# scrapy.responsetypes.responsetypes.from_args
PRE_XTRACTMIME_SCENARIOS = (
(
{
'url': 'http://www.example.com/data.csv',
},
TextResponse,
),
(
{
'url': 'http://www.example.com/item/',
'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}),
},
HtmlResponse,
),
(
{
'url': 'http://www.example.com/page/',
'headers': Headers(
{
'Content-Disposition': [
'attachment; filename="data.xml.gz"',
]
}
),
'body': b'\x01\x02',
},
Response,
),
(
{'body': b'Some plain\0 text data with\0 tabs and null bytes\0'},
TextResponse,
),
(
{
'body': b'\x03\x02\xdf\xdd\x23',
'headers': Headers({'Content-Encoding': 'UTF-8'}),
},
Response,
),
(
{
'body': b'\x00\x01\xff',
'url': '://www.example.com/item/',
'headers': Headers({'Content-Type': ['text/plain']}),
},
TextResponse,
),
({'url': 'http://www.example.com/item/file.html'}, HtmlResponse),
({'body': b'
Hello'}, HtmlResponse),
({'body': b'\n.'}, HtmlResponse),
(
{
'body': b'\x01\x02',
'headers': Headers({'Content-Type': ['application/pdf']}),
},
Response,
),
(
{'headers': Headers({'Content-Type': ['application/x-json']})},
TextResponse,
),
(
{'headers': Headers({'Content-Type': ['application/x-javascript']})},
TextResponse,
),
(
{
'headers': Headers(
{'Content-Type': ['application/json-amazonui-streaming']}
)
},
TextResponse,
),
(
{
'headers': Headers(
{'Content-Disposition': ['attachment; filename="data.xml.gz"']}
),
'url': 'http://www.example.com/page/',
},
Response,
),
({'headers': Headers({'Content-Type': ['application/pdf']})}, Response),
(
{
'url': 'http://www.example.com/page/file.html',
'headers': Headers({'Content-Type': 'application/octet-stream'}),
},
HtmlResponse,
),
(
{
'url': 'http://www.example.com/item/file.xml',
'headers': Headers({'Content-Type': 'application/octet-stream'}),
},
XmlResponse,
),
(
{
'url': 'http://www.example.com/item/file.html',
'headers': Headers({'Content-Type': 'application/octet-stream'}),
},
HtmlResponse,
),
(
{
'url': 'http://www.example.com/item/file.xml',
'headers': Headers(
{
'Content-Disposition': [
'attachment; filename="data.xml.gz"'
],
'Content-Type': 'application/octet-stream',
}
),
},
XmlResponse,
),
({'url': 'http://www.example.com/item/file.pdf'}, Response),
({'filename': 'file.pdf'}, Response),
(
{
'body': b'\n.',
'url': 'http://www.example.com',
},
HtmlResponse,
),
)
# Scenarios that work differently with the previously-used, deprecated
# scrapy.responsetypes.responsetypes.from_args
POST_XTRACTMIME_SCENARIOS = (
(
{
'body': b'\x00\x01\xff',
'url': 'http://www.example.com/item/',
'headers': Headers({'Content-Type': ['text/plain']}),
},
Response,
),
({'filename': '/tmp/temp^'}, TextResponse),
({'body': b'%PDF-1.4'}, Response),
(
{'headers': Headers({'Content-Type': ['application/ecmascript']})},
TextResponse,
),
(
{'headers': Headers({'Content-Type': ['application/ld+json']})},
TextResponse,
),
(
{
'headers': Headers(
{'Content-Encoding': ['zip'], 'Content-Type': ['text/html']}
)
},
HtmlResponse,
),
(
{
'headers': Headers(
{'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']}
)
},
TextResponse,
),
(
{
'body': b'Non HTML',
'headers': Headers(
{'Content-Encoding': ['zip'], 'Content-Type': ['text/html']}
),
},
HtmlResponse,
),
(
{
'body': b'Some plain text',
'headers': Headers({'Content-Type': 'application/octet-stream'}),
},
Response,
),
({'body': b'\x0c\x1b'}, TextResponse),
({'body': b'this is not '}, TextResponse),
({'body': b'this is not test page test body "
def browser_open(burl):
path = urlparse(burl).path
if not os.path.exists(path):
path = burl.replace('file://', '')
with open(path, "rb") as f:
bbody = f.read()
self.assertIn(b'', bbody)
return True
response = HtmlResponse(url, body=body)
assert open_in_browser(response, _openfunc=browser_open), "Browser not called"
resp = Response(url, body=body)
self.assertRaises(TypeError, open_in_browser, resp, debug=True)
def test_get_meta_refresh(self):
r1 = HtmlResponse("http://www.example.com", body=b"""
Dummy
blahablsdfsal&
""")
r2 = HtmlResponse("http://www.example.com", body=b"""
Dummy
blahablsdfsal&
""")
r3 = HtmlResponse("http://www.example.com", body=b"""