mirror of https://github.com/scrapy/scrapy.git
Add verbatim_url
This commit is contained in:
parent
b9c2240040
commit
4a93f26746
|
|
@ -117,6 +117,9 @@ Request objects
|
|||
:param encoding: the encoding of this request (defaults to ``'utf-8'``).
|
||||
This encoding will be used to percent-encode the URL and to convert the
|
||||
body to bytes (if given as a string).
|
||||
|
||||
To disable URL percent-encoding for a request, use the
|
||||
:reqmeta:`verbatim_url` request meta key.
|
||||
:type encoding: str
|
||||
|
||||
:param priority: sets :attr:`priority`, defaults to ``0``.
|
||||
|
|
@ -136,9 +139,13 @@ Request objects
|
|||
|
||||
.. attribute:: Request.url
|
||||
|
||||
A string containing the URL of this request. Keep in mind that this
|
||||
attribute contains the escaped URL, so it can differ from the URL passed in
|
||||
the ``__init__()`` method.
|
||||
A string containing the URL of this request.
|
||||
|
||||
Keep in mind that this attribute contains the escaped URL, so it can
|
||||
differ from the URL passed in the ``__init__()`` method.
|
||||
|
||||
If :reqmeta:`verbatim_url` is set to ``True``, the URL is kept as
|
||||
passed to ``__init__()``.
|
||||
|
||||
This attribute is read-only. To change the URL of a Request use
|
||||
:meth:`replace`.
|
||||
|
|
@ -469,6 +476,11 @@ in your :meth:`fingerprint` method implementation:
|
|||
|
||||
.. autofunction:: scrapy.utils.request.fingerprint
|
||||
|
||||
By default, request fingerprinting canonicalizes the request URL. If
|
||||
:reqmeta:`verbatim_url` is set to ``True``, fingerprinting does not
|
||||
canonicalize the URL, and the ``keep_fragments`` parameter is ignored (it is
|
||||
effectively true).
|
||||
|
||||
For example, to take the value of a request header named ``X-ID`` into
|
||||
account:
|
||||
|
||||
|
|
@ -638,6 +650,7 @@ Those are:
|
|||
* :reqmeta:`redirect_reasons`
|
||||
* :reqmeta:`redirect_urls`
|
||||
* :reqmeta:`referrer_policy`
|
||||
* :reqmeta:`verbatim_url`
|
||||
|
||||
.. reqmeta:: bindaddress
|
||||
|
||||
|
|
@ -714,6 +727,21 @@ The meta key is used set retry times per request. When initialized, the
|
|||
:reqmeta:`max_retry_times` meta key takes higher precedence over the
|
||||
:setting:`RETRY_TIMES` setting.
|
||||
|
||||
.. reqmeta:: verbatim_url
|
||||
|
||||
verbatim_url
|
||||
------------
|
||||
|
||||
Set this key to ``True`` to keep the request URL as passed to
|
||||
:class:`~scrapy.Request`, without URL percent-encoding.
|
||||
|
||||
When this key is enabled, :func:`~scrapy.utils.request.fingerprint` does not
|
||||
canonicalize the request URL, so requests whose URLs differ only in
|
||||
characters that would otherwise be canonicalized get different fingerprints.
|
||||
|
||||
In this mode, the ``keep_fragments`` parameter is ignored, and it is
|
||||
effectively true.
|
||||
|
||||
|
||||
.. _topics-stop-response-download:
|
||||
|
||||
|
|
|
|||
|
|
@ -138,6 +138,7 @@ class Request(object_ref):
|
|||
) -> None:
|
||||
self._encoding: str = encoding # this one has to be set first
|
||||
self.method: str = str(method).upper()
|
||||
self._meta: dict[str, Any] | None = dict(meta) if meta else None
|
||||
self._set_url(url)
|
||||
self._set_body(body)
|
||||
if not isinstance(priority, int):
|
||||
|
|
@ -232,7 +233,6 @@ class Request(object_ref):
|
|||
#: default. See :meth:`~scrapy.Spider.start`.
|
||||
self.dont_filter: bool = dont_filter
|
||||
|
||||
self._meta: dict[str, Any] | None = dict(meta) if meta else None
|
||||
self._cb_kwargs: dict[str, Any] | None = dict(cb_kwargs) if cb_kwargs else None
|
||||
self._flags: list[str] | None = list(flags) if flags else None
|
||||
|
||||
|
|
@ -252,11 +252,17 @@ class Request(object_ref):
|
|||
def url(self) -> str:
|
||||
return self._url
|
||||
|
||||
def _url_is_verbatim(self) -> bool:
|
||||
return bool(self._meta and self._meta.get("verbatim_url"))
|
||||
|
||||
def _set_url(self, url: str) -> None:
|
||||
if not isinstance(url, str):
|
||||
raise TypeError(f"Request url must be str, got {type(url).__name__}")
|
||||
|
||||
self._url = safe_url_string(url, self.encoding)
|
||||
if self._url_is_verbatim():
|
||||
self._url = url
|
||||
else:
|
||||
self._url = safe_url_string(url, self.encoding)
|
||||
|
||||
if (
|
||||
"://" not in self._url
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ if TYPE_CHECKING:
|
|||
|
||||
|
||||
_fingerprint_cache: WeakKeyDictionary[
|
||||
Request, dict[tuple[tuple[bytes, ...] | None, bool], bytes]
|
||||
Request, dict[tuple[tuple[bytes, ...] | None, bool, bool], bytes]
|
||||
] = WeakKeyDictionary()
|
||||
|
||||
|
||||
|
|
@ -71,8 +71,10 @@ def fingerprint(
|
|||
processed_include_headers = tuple(
|
||||
to_bytes(h.lower()) for h in sorted(include_headers)
|
||||
)
|
||||
verbatim_url = bool(request.meta.get("verbatim_url"))
|
||||
effective_keep_fragments = keep_fragments and not verbatim_url
|
||||
cache = _fingerprint_cache.setdefault(request, {})
|
||||
cache_key = (processed_include_headers, keep_fragments)
|
||||
cache_key = (processed_include_headers, effective_keep_fragments, verbatim_url)
|
||||
if cache_key not in cache:
|
||||
# To decode bytes reliably (JSON does not support bytes), regardless of
|
||||
# character encoding, we use bytes.hex()
|
||||
|
|
@ -84,9 +86,13 @@ def fingerprint(
|
|||
header_value.hex()
|
||||
for header_value in request.headers.getlist(header)
|
||||
]
|
||||
if verbatim_url:
|
||||
url = request.url
|
||||
else:
|
||||
url = canonicalize_url(request.url, keep_fragments=keep_fragments)
|
||||
fingerprint_data = {
|
||||
"method": to_unicode(request.method),
|
||||
"url": canonicalize_url(request.url, keep_fragments=keep_fragments),
|
||||
"url": url,
|
||||
"body": (request.body or b"").hex(),
|
||||
"headers": headers,
|
||||
}
|
||||
|
|
@ -108,8 +114,9 @@ class RequestFingerprinter:
|
|||
(:func:`w3lib.url.canonicalize_url`) of :attr:`request.url
|
||||
<scrapy.Request.url>` and the values of :attr:`request.method
|
||||
<scrapy.Request.method>` and :attr:`request.body
|
||||
<scrapy.Request.body>`. It then generates an `SHA1
|
||||
<https://en.wikipedia.org/wiki/SHA-1>`_ hash.
|
||||
<scrapy.Request.body>`, unless :reqmeta:`verbatim_url` is true for that
|
||||
request. It then generates an `SHA1 <https://en.wikipedia.org/wiki/SHA-1>`_
|
||||
hash.
|
||||
"""
|
||||
|
||||
@classmethod
|
||||
|
|
|
|||
|
|
@ -197,6 +197,8 @@ class TestHttp2WithCrawler(TestHttpWithCrawlerBase):
|
|||
class TestHttps2Proxy(H2DownloadHandlerMixin, TestHttpProxyBase):
|
||||
is_secure = True
|
||||
expected_http_proxy_request_body = b"/"
|
||||
expected_http_proxy_quoted_request_body = b"/list?%5B0%5D=a"
|
||||
expected_http_proxy_verbatim_request_body = b"/list?[0]=a"
|
||||
|
||||
@coroutine_test
|
||||
async def test_download_with_proxy_https_timeout(
|
||||
|
|
|
|||
|
|
@ -1075,6 +1075,8 @@ class TestHttpWithCrawlerBase(ABC):
|
|||
class TestHttpProxyBase(ABC):
|
||||
is_secure = False
|
||||
expected_http_proxy_request_body = b"http://example.com"
|
||||
expected_http_proxy_quoted_request_body = b"http://example.com/list?%5B0%5D=a"
|
||||
expected_http_proxy_verbatim_request_body = b"http://example.com/list?[0]=a"
|
||||
|
||||
@property
|
||||
@abstractmethod
|
||||
|
|
@ -1146,3 +1148,25 @@ class TestHttpProxyBase(ABC):
|
|||
assert response.status == 200
|
||||
assert response.url == request.url
|
||||
assert response.body == self.expected_http_proxy_request_body
|
||||
|
||||
@coroutine_test
|
||||
async def test_download_with_proxy_verbatim_url(
|
||||
self, proxy_mockserver: ProxyEchoMockServer
|
||||
) -> None:
|
||||
http_proxy = proxy_mockserver.url("", is_secure=self.is_secure)
|
||||
url = "http://example.com/list?[0]=a"
|
||||
request = Request(url, meta={"proxy": http_proxy})
|
||||
request_verbatim = Request(
|
||||
url, meta={"proxy": http_proxy, "verbatim_url": True}
|
||||
)
|
||||
|
||||
async with self.get_dh() as download_handler:
|
||||
response = await download_handler.download_request(request)
|
||||
response_verbatim = await download_handler.download_request(
|
||||
request_verbatim
|
||||
)
|
||||
|
||||
assert response.status == 200
|
||||
assert response_verbatim.status == 200
|
||||
assert response.body == self.expected_http_proxy_quoted_request_body
|
||||
assert response_verbatim.body == self.expected_http_proxy_verbatim_request_body
|
||||
|
|
|
|||
|
|
@ -103,6 +103,14 @@ class TestRequest:
|
|||
r = self.request_class(url="http://www.scrapy.org/blank space")
|
||||
assert r.url == "http://www.scrapy.org/blank%20space"
|
||||
|
||||
def test_url_verbatim_meta(self):
|
||||
url = "http://www.scrapy.org/list?[0]=a"
|
||||
r = self.request_class(url=url)
|
||||
assert r.url == "http://www.scrapy.org/list?%5B0%5D=a"
|
||||
|
||||
r = self.request_class(url=url, meta={"verbatim_url": True})
|
||||
assert r.url == url
|
||||
|
||||
def test_url_encoding(self):
|
||||
r = self.request_class(url="http://www.scrapy.org/price/£")
|
||||
assert r.url == "http://www.scrapy.org/price/%C2%A3"
|
||||
|
|
|
|||
|
|
@ -58,10 +58,14 @@ def test_request_httprepr_for_non_http_request(r: Request) -> None:
|
|||
class TestFingerprint:
|
||||
function: staticmethod = staticmethod(fingerprint)
|
||||
cache: (
|
||||
WeakKeyDictionary[Request, dict[tuple[tuple[bytes, ...] | None, bool], bytes]]
|
||||
| WeakKeyDictionary[Request, dict[tuple[tuple[bytes, ...] | None, bool], str]]
|
||||
WeakKeyDictionary[
|
||||
Request, dict[tuple[tuple[bytes, ...] | None, bool, bool], bytes]
|
||||
]
|
||||
| WeakKeyDictionary[
|
||||
Request, dict[tuple[tuple[bytes, ...] | None, bool, bool], str]
|
||||
]
|
||||
) = _fingerprint_cache
|
||||
default_cache_key = (None, False)
|
||||
default_cache_key = (None, False, False)
|
||||
known_hashes: tuple[tuple[Request, bytes | str, dict], ...] = (
|
||||
(
|
||||
Request("http://example.org"),
|
||||
|
|
@ -128,6 +132,16 @@ class TestFingerprint:
|
|||
b"\xc1\xef~\x94\x9bS\xc1\x83\t\xdcz8\x9f\xdc{\x11\x16I.\x11",
|
||||
{"include_headers": ["A"], "keep_fragments": True},
|
||||
),
|
||||
(
|
||||
Request("https://example.org/#a", meta={"verbatim_url": True}),
|
||||
b"<\x1a\xeb\x85y\xdeW\xfb\xdcq\x88\xee\xaf\x17\xdd\x0c\xbfH\x18\x1f",
|
||||
{},
|
||||
),
|
||||
(
|
||||
Request("https://example.org/#a", meta={"verbatim_url": True}),
|
||||
b"<\x1a\xeb\x85y\xdeW\xfb\xdcq\x88\xee\xaf\x17\xdd\x0c\xbfH\x18\x1f",
|
||||
{"keep_fragments": True},
|
||||
),
|
||||
(
|
||||
Request("https://example.org/ab"),
|
||||
b"N\xe5l\xb8\x12@iw\xe2\xf3\x1bp\xea\xffp!u\xe2\x8a\xc6",
|
||||
|
|
@ -146,6 +160,21 @@ class TestFingerprint:
|
|||
assert self.function(r1) == self.function(r1)
|
||||
assert self.function(r1) == self.function(r2)
|
||||
|
||||
def test_verbatim_url(self):
|
||||
raw_url = "https://example.org/?id[]=1"
|
||||
canonical_url = "https://example.org/?id%5B%5D=1"
|
||||
|
||||
request = Request(raw_url)
|
||||
request_verbatim_raw = Request(raw_url, meta={"verbatim_url": True})
|
||||
request_verbatim_canonical = Request(canonical_url, meta={"verbatim_url": True})
|
||||
|
||||
assert request.url == canonical_url
|
||||
assert self.function(request) == self.function(request_verbatim_canonical)
|
||||
assert self.function(request_verbatim_raw) != self.function(request)
|
||||
assert self.function(request_verbatim_raw) != self.function(
|
||||
request_verbatim_canonical
|
||||
)
|
||||
|
||||
def test_query_string_key_without_value(self):
|
||||
r1 = Request("http://www.example.com/hnnoticiaj1.aspx?78132,199")
|
||||
r2 = Request("http://www.example.com/hnnoticiaj1.aspx?78160,199")
|
||||
|
|
@ -191,6 +220,23 @@ class TestFingerprint:
|
|||
assert self.function(r2) != self.function(r2, keep_fragments=True)
|
||||
assert self.function(r1) != self.function(r2, keep_fragments=True)
|
||||
|
||||
def test_verbatim_url_fragment(self):
|
||||
r1 = Request(
|
||||
"http://www.example.com/test.html#fragment1",
|
||||
meta={"verbatim_url": True},
|
||||
)
|
||||
r2 = Request(
|
||||
"http://www.example.com/test.html#fragment2",
|
||||
meta={"verbatim_url": True},
|
||||
)
|
||||
|
||||
assert self.function(r1) != self.function(r2)
|
||||
assert self.function(r1) == self.function(r1, keep_fragments=True)
|
||||
assert self.function(r2) == self.function(r2, keep_fragments=True)
|
||||
assert self.function(r1, keep_fragments=True) != self.function(
|
||||
r2, keep_fragments=True
|
||||
)
|
||||
|
||||
def test_method_and_body(self):
|
||||
r1 = Request("http://www.example.com")
|
||||
r2 = Request("http://www.example.com", method="POST")
|
||||
|
|
|
|||
Loading…
Reference in New Issue