mirror of https://github.com/scrapy/scrapy.git
Merge c9bb3a4f4b into d8ba1571e7
This commit is contained in:
commit
36cd110614
|
|
@ -298,7 +298,7 @@ Does Scrapy manage cookies automatically?
|
||||||
Yes, Scrapy receives and keeps track of cookies sent by servers, and sends them
|
Yes, Scrapy receives and keeps track of cookies sent by servers, and sends them
|
||||||
back on subsequent requests, like any regular web browser does.
|
back on subsequent requests, like any regular web browser does.
|
||||||
|
|
||||||
For more info see :ref:`topics-request-response` and :ref:`cookies-mw`.
|
For more info see :ref:`topics-request-response`, :ref:`cookies-mw` and :ref:`cookiejars`.
|
||||||
|
|
||||||
How can I see the cookies being sent and received from Scrapy?
|
How can I see the cookies being sent and received from Scrapy?
|
||||||
--------------------------------------------------------------
|
--------------------------------------------------------------
|
||||||
|
|
|
||||||
|
|
@ -271,6 +271,94 @@ Here's an example of a log with :setting:`COOKIES_DEBUG` enabled::
|
||||||
[...]
|
[...]
|
||||||
|
|
||||||
|
|
||||||
|
.. _cookiejars:
|
||||||
|
|
||||||
|
Direct access to cookiejars from spider
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
|
||||||
|
In some cases it is required to directly set specific values in an existing :class:`scrapy.http.cookies.CookieJar` object where :class:`http.cookiejar.CookieJar` from python standard library used here)
|
||||||
|
|
||||||
|
.. skip: start
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
import scrapy
|
||||||
|
from scrapy.crawler import CrawlerProcess
|
||||||
|
|
||||||
|
|
||||||
|
class Quotes(scrapy.Spider):
|
||||||
|
name = "quotes"
|
||||||
|
custom_settings = {"DOWNLOAD_DELAY": 1}
|
||||||
|
|
||||||
|
def start_requests(self):
|
||||||
|
yield scrapy.Request(
|
||||||
|
url="https://quotes.toscrape.com/login", callback=self.login
|
||||||
|
)
|
||||||
|
|
||||||
|
def login(self, response):
|
||||||
|
self.logger.info(self.cookie_jars[None]) # scrapy.http.cookies.CookieJar object
|
||||||
|
self.logger.info(self.cookie_jars[None].jar) # http.cookiejar.CookieJar object
|
||||||
|
|
||||||
|
locale_cookie = (
|
||||||
|
self.cookie_jars[None]._cookies["quotes.toscrape.com"]["/"].get("session")
|
||||||
|
)
|
||||||
|
locale_cookie.value = locale_cookie.value.upper()
|
||||||
|
self.logger.info(self.cookie_jars[None].jar)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
p = CrawlerProcess()
|
||||||
|
p.crawl(Quotes)
|
||||||
|
p.start()
|
||||||
|
|
||||||
|
.. skip: end
|
||||||
|
|
||||||
|
.. note::
|
||||||
|
|
||||||
|
If :reqmeta:`cookiejar` didn't specified in request meta, middleware will read this as ``None``, as result of this middleware will create cookiejar object under dict key ``None``.
|
||||||
|
|
||||||
|
|
||||||
|
Log output::
|
||||||
|
|
||||||
|
2024-02-23 10:51:27 [scrapy.core.engine] DEBUG: Crawled (200) <GET https://quotes.toscrape.com/login> (referer: None)
|
||||||
|
2024-02-23 10:51:27 [quotes] INFO: <scrapy.http.cookies.CookieJar object at 0x00000217DB719B40>
|
||||||
|
2024-02-23 10:51:27 [quotes] INFO: <CookieJar[<Cookie session=eyJjc3JmX3Rva2VuIjoiSnFQQU9GTGt1amZzZ3J3UVdHeGV6WFR2UnBpY0Job1NWS3liWmxhblVISXROREVtQ2RZTSJ9.Zdhqng.8uQzjuvDfOcNJHV7luY5Na6C1N0 for quotes.toscrape.com/>]>
|
||||||
|
2024-02-23 10:51:27 [quotes] INFO: <CookieJar[<Cookie session=EYJJC3JMX3RVA2VUIJOISNFQQU9GTGT1AMZZZ3J3UVDHEGV6WFR2UNBPY0JOB1NWS3LIWMXHBLVISXROREVTQ2RZTSJ9.ZDHQNG.8UQZJUVDFOCNJHV7LUY5NA6C1N0 for quotes.toscrape.com/>]>
|
||||||
|
2024-02-23 10:51:27 [scrapy.core.engine] INFO: Closing spider (finished)
|
||||||
|
|
||||||
|
|
||||||
|
If cookie middleware is enabled, ``get_cookiejar(response_or_request)`` method added to spider:
|
||||||
|
|
||||||
|
.. class:: scrapy.Spider
|
||||||
|
|
||||||
|
...
|
||||||
|
|
||||||
|
.. method:: get_cookiejar(spider, response_or_request)
|
||||||
|
|
||||||
|
Returns ``scrapy.http.cookies.CookieJar`` cookiejar object based on :reqmeta:`cookiejar` value of ``response_or_request`` argument
|
||||||
|
|
||||||
|
:param response_or_request: request or response object where target session/cookiejar object used or planned to use
|
||||||
|
:type response_or_request: :class:`~scrapy.Request` or :class:`~scrapy.Response` object
|
||||||
|
|
||||||
|
Technically this is short cut to ``request.meta.get('cookiejar')`` or ``response.request.meta.get('cookiejar')`` if ``response_or_request`` is :class:`~scrapy.Response`
|
||||||
|
|
||||||
|
It allows a bit easier access to cookiejar object:
|
||||||
|
|
||||||
|
|
||||||
|
.. code-block:: none
|
||||||
|
|
||||||
|
self.logger.info(self.get_cookiejar(response)) # scrapy.http.cookies.CookieJar object
|
||||||
|
self.logger.info(self.get_cookiejar(response).jar) # http.cookiejar.CookieJar object
|
||||||
|
locale_cookie = (
|
||||||
|
self.get_cookiejar(response).jar._cookies["quotes.toscrape.com"]["/"].get("session")
|
||||||
|
)
|
||||||
|
|
||||||
|
Known use cases where direct access to cookiejar objects can be useful:
|
||||||
|
|
||||||
|
1. Modifying specific cookie value (as provided on example code sample above).
|
||||||
|
2. Delete cookiejar. On some cases this can re-initialise session.
|
||||||
|
|
||||||
|
.. warning:: ``http.cookiejar.CookieJar`` object from python standard lib https://docs.python.org/3/library/http.cookies.html doesn't provide option to individually access/modify specific cookie values. It is possible to change it only using it's private attributes as provided on code sample above.
|
||||||
|
|
||||||
DefaultHeadersMiddleware
|
DefaultHeadersMiddleware
|
||||||
------------------------
|
------------------------
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -6,6 +6,7 @@ from typing import TYPE_CHECKING, Any
|
||||||
|
|
||||||
from tldextract import TLDExtract
|
from tldextract import TLDExtract
|
||||||
|
|
||||||
|
from scrapy import signals
|
||||||
from scrapy.exceptions import NotConfigured
|
from scrapy.exceptions import NotConfigured
|
||||||
from scrapy.http import Response
|
from scrapy.http import Response
|
||||||
from scrapy.http.cookies import CookieJar
|
from scrapy.http.cookies import CookieJar
|
||||||
|
|
@ -38,6 +39,14 @@ def _is_public_domain(domain: str) -> bool:
|
||||||
return not parts.domain
|
return not parts.domain
|
||||||
|
|
||||||
|
|
||||||
|
def get_cookiejar(self, response_or_request):
|
||||||
|
try:
|
||||||
|
cookiejarkey = response_or_request.request.meta.get("cookiejar")
|
||||||
|
except Exception:
|
||||||
|
cookiejarkey = response_or_request.meta.get("cookiejar")
|
||||||
|
return self.cookie_jars[cookiejarkey]
|
||||||
|
|
||||||
|
|
||||||
class CookiesMiddleware:
|
class CookiesMiddleware:
|
||||||
"""This middleware enables working with sites that need cookies"""
|
"""This middleware enables working with sites that need cookies"""
|
||||||
|
|
||||||
|
|
@ -53,8 +62,14 @@ class CookiesMiddleware:
|
||||||
raise NotConfigured
|
raise NotConfigured
|
||||||
o = cls(crawler.settings.getbool("COOKIES_DEBUG"))
|
o = cls(crawler.settings.getbool("COOKIES_DEBUG"))
|
||||||
o.crawler = crawler
|
o.crawler = crawler
|
||||||
|
crawler.signals.connect(o.spider_opened, signal=signals.spider_opened)
|
||||||
return o
|
return o
|
||||||
|
|
||||||
|
def spider_opened(self, spider: Spider) -> None:
|
||||||
|
spider.cookie_jars = self.jars
|
||||||
|
attribute_name = "get_cookiejar"
|
||||||
|
setattr(spider.__class__, attribute_name, get_cookiejar)
|
||||||
|
|
||||||
def _process_cookies(
|
def _process_cookies(
|
||||||
self, cookies: Iterable[Cookie], *, jar: CookieJar, request: Request
|
self, cookies: Iterable[Cookie], *, jar: CookieJar, request: Request
|
||||||
) -> None:
|
) -> None:
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue