Merge pull request #6335 from wRAR/typing-self

Use the Self type hint in from_crawler/from_settings.
This commit is contained in:
Andrey Rakhmatullin 2024-05-06 14:02:10 +04:00 committed by GitHub
commit d338982580
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
14 changed files with 130 additions and 36 deletions

View File

@ -4,3 +4,9 @@ include = scrapy/*
omit =
tests/*
disable_warnings = include-ignored
[report]
# https://github.com/nedbat/coveragepy/issues/831#issuecomment-517778185
exclude_lines =
pragma: no cover
if TYPE_CHECKING:

View File

@ -1,3 +1,5 @@
from __future__ import annotations
import warnings
from typing import TYPE_CHECKING, Any, List, Optional
@ -25,6 +27,9 @@ from scrapy.utils.misc import build_from_crawler, load_object
if TYPE_CHECKING:
from twisted.internet._sslverify import ClientTLSOptions
# typing.Self requires Python 3.11
from typing_extensions import Self
@implementer(IPolicyForHTTPS)
class ScrapyClientContextFactory(BrowserLikePolicyForHTTPS):
@ -62,7 +67,7 @@ class ScrapyClientContextFactory(BrowserLikePolicyForHTTPS):
method: int = SSL.SSLv23_METHOD,
*args: Any,
**kwargs: Any,
):
) -> Self:
tls_verbose_logging: bool = settings.getbool(
"DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING"
)

View File

@ -28,18 +28,26 @@ In case of status 200 request, response.headers will come with two keys:
'Size' - with size of the downloaded data
"""
from __future__ import annotations
import re
from io import BytesIO
from typing import TYPE_CHECKING
from urllib.parse import unquote
from twisted.internet.protocol import ClientCreator, Protocol
from twisted.protocols.ftp import CommandFailed, FTPClient
from scrapy.crawler import Crawler
from scrapy.http import Response
from scrapy.responsetypes import responsetypes
from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.python import to_bytes
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
class ReceivedDataProtocol(Protocol):
def __init__(self, filename=None):
@ -76,7 +84,7 @@ class FTPDownloadHandler:
self.passive_mode = settings["FTP_PASSIVE_MODE"]
@classmethod
def from_crawler(cls, crawler):
def from_crawler(cls, crawler: Crawler) -> Self:
return cls(crawler.settings)
def download_request(self, request, spider):

View File

@ -1,9 +1,17 @@
"""Download handlers for http and https schemes
"""
from __future__ import annotations
from typing import TYPE_CHECKING
from scrapy.utils.misc import build_from_crawler, load_object
from scrapy.utils.python import to_unicode
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
class HTTP10DownloadHandler:
lazy = False
@ -17,7 +25,7 @@ class HTTP10DownloadHandler:
self._crawler = crawler
@classmethod
def from_crawler(cls, crawler):
def from_crawler(cls, crawler) -> Self:
return cls(crawler.settings, crawler)
def download_request(self, request, spider):

View File

@ -1,11 +1,14 @@
"""Download handlers for http and https schemes"""
from __future__ import annotations
import ipaddress
import logging
import re
from contextlib import suppress
from io import BytesIO
from time import time
from typing import TYPE_CHECKING
from urllib.parse import urldefrag, urlunparse
from twisted.internet import defer, protocol, ssl
@ -32,6 +35,11 @@ from scrapy.http import Headers
from scrapy.responsetypes import responsetypes
from scrapy.utils.python import to_bytes, to_unicode
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
logger = logging.getLogger(__name__)
@ -56,7 +64,7 @@ class HTTP11DownloadHandler:
self._disconnect_timeout = 1
@classmethod
def from_crawler(cls, crawler):
def from_crawler(cls, crawler) -> Self:
return cls(crawler.settings, crawler)
def download_request(self, request, spider):

View File

@ -1,5 +1,7 @@
from __future__ import annotations
from time import time
from typing import Optional, Type, TypeVar
from typing import TYPE_CHECKING, Optional
from urllib.parse import urldefrag
from twisted.internet.base import DelayedCall
@ -16,9 +18,9 @@ from scrapy.settings import Settings
from scrapy.spiders import Spider
from scrapy.utils.python import to_bytes
H2DownloadHandlerOrSubclass = TypeVar(
"H2DownloadHandlerOrSubclass", bound="H2DownloadHandler"
)
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
class H2DownloadHandler:
@ -31,9 +33,7 @@ class H2DownloadHandler:
self._context_factory = load_context_factory_from_settings(settings, crawler)
@classmethod
def from_crawler(
cls: Type[H2DownloadHandlerOrSubclass], crawler: Crawler
) -> H2DownloadHandlerOrSubclass:
def from_crawler(cls, crawler: Crawler) -> Self:
return cls(crawler.settings, crawler)
def download_request(self, request: Request, spider: Spider) -> Deferred:

View File

@ -1,9 +1,17 @@
from __future__ import annotations
from typing import TYPE_CHECKING
from scrapy.core.downloader.handlers.http import HTTPDownloadHandler
from scrapy.exceptions import NotConfigured
from scrapy.utils.boto import is_botocore_available
from scrapy.utils.httpobj import urlparse_cached
from scrapy.utils.misc import build_from_crawler
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
class S3DownloadHandler:
def __init__(
@ -57,7 +65,7 @@ class S3DownloadHandler:
self._download_http = _http_handler.download_request
@classmethod
def from_crawler(cls, crawler, **kwargs):
def from_crawler(cls, crawler, **kwargs) -> Self:
return cls(crawler.settings, crawler=crawler, **kwargs)
def download_request(self, request, spider):

View File

@ -5,8 +5,11 @@ requests in Scrapy.
See documentation in docs/topics/request-response.rst
"""
from __future__ import annotations
import inspect
from typing import (
TYPE_CHECKING,
Any,
AnyStr,
Callable,
@ -17,8 +20,6 @@ from typing import (
NoReturn,
Optional,
Tuple,
Type,
TypeVar,
Union,
cast,
)
@ -32,7 +33,9 @@ from scrapy.utils.python import to_bytes
from scrapy.utils.trackref import object_ref
from scrapy.utils.url import escape_ajax
RequestTypeVar = TypeVar("RequestTypeVar", bound="Request")
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
def NO_CALLBACK(*args: Any, **kwargs: Any) -> NoReturn:
@ -186,11 +189,11 @@ class Request(object_ref):
@classmethod
def from_curl(
cls: Type[RequestTypeVar],
cls,
curl_command: str,
ignore_unknown_options: bool = True,
**kwargs: Any,
) -> RequestTypeVar:
) -> Self:
"""Create a Request object from a string containing a `cURL
<https://curl.haxx.se/>`_ command. It populates the HTTP method, the
URL, the headers, the cookies and the body. It accepts the same

View File

@ -4,6 +4,8 @@ Files Pipeline
See documentation in topics/media-pipeline.rst
"""
from __future__ import annotations
import base64
import functools
import hashlib
@ -16,7 +18,7 @@ from ftplib import FTP
from io import BytesIO
from os import PathLike
from pathlib import Path
from typing import IO, DefaultDict, Optional, Set, Union
from typing import IO, TYPE_CHECKING, DefaultDict, Optional, Set, Type, Union, cast
from urllib.parse import urlparse
from itemadapter import ItemAdapter
@ -34,6 +36,10 @@ from scrapy.utils.log import failure_to_exc_info
from scrapy.utils.python import to_bytes
from scrapy.utils.request import referer_str
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
logger = logging.getLogger(__name__)
@ -385,8 +391,8 @@ class FilesPipeline(MediaPipeline):
super().__init__(download_func=download_func, settings=settings)
@classmethod
def from_settings(cls, settings):
s3store = cls.STORE_SCHEMES["s3"]
def from_settings(cls, settings) -> Self:
s3store: Type[S3FilesStore] = cast(Type[S3FilesStore], cls.STORE_SCHEMES["s3"])
s3store.AWS_ACCESS_KEY_ID = settings["AWS_ACCESS_KEY_ID"]
s3store.AWS_SECRET_ACCESS_KEY = settings["AWS_SECRET_ACCESS_KEY"]
s3store.AWS_SESSION_TOKEN = settings["AWS_SESSION_TOKEN"]
@ -396,11 +402,15 @@ class FilesPipeline(MediaPipeline):
s3store.AWS_VERIFY = settings["AWS_VERIFY"]
s3store.POLICY = settings["FILES_STORE_S3_ACL"]
gcs_store = cls.STORE_SCHEMES["gs"]
gcs_store: Type[GCSFilesStore] = cast(
Type[GCSFilesStore], cls.STORE_SCHEMES["gs"]
)
gcs_store.GCS_PROJECT_ID = settings["GCS_PROJECT_ID"]
gcs_store.POLICY = settings["FILES_STORE_GCS_ACL"] or None
ftp_store = cls.STORE_SCHEMES["ftp"]
ftp_store: Type[FTPFilesStore] = cast(
Type[FTPFilesStore], cls.STORE_SCHEMES["ftp"]
)
ftp_store.FTP_USERNAME = settings["FTP_USER"]
ftp_store.FTP_PASSWORD = settings["FTP_PASSWORD"]
ftp_store.USE_ACTIVE_MODE = settings.getbool("FEED_STORAGE_FTP_ACTIVE")

View File

@ -4,25 +4,38 @@ Images Pipeline
See documentation in topics/media-pipeline.rst
"""
from __future__ import annotations
import functools
import hashlib
import warnings
from contextlib import suppress
from io import BytesIO
from os import PathLike
from typing import Dict, Tuple, Union
from typing import TYPE_CHECKING, Dict, Tuple, Type, Union, cast
from itemadapter import ItemAdapter
from scrapy.exceptions import DropItem, NotConfigured, ScrapyDeprecationWarning
from scrapy.http import Request
from scrapy.http.request import NO_CALLBACK
from scrapy.pipelines.files import FileException, FilesPipeline, _md5sum
from scrapy.pipelines.files import (
FileException,
FilesPipeline,
FTPFilesStore,
GCSFilesStore,
S3FilesStore,
_md5sum,
)
# TODO: from scrapy.pipelines.media import MediaPipeline
from scrapy.settings import Settings
from scrapy.utils.python import get_func_args, to_bytes
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
class NoimagesDrop(DropItem):
"""Product with no images exception"""
@ -96,8 +109,8 @@ class ImagesPipeline(FilesPipeline):
self._deprecated_convert_image = None
@classmethod
def from_settings(cls, settings):
s3store = cls.STORE_SCHEMES["s3"]
def from_settings(cls, settings) -> Self:
s3store: Type[S3FilesStore] = cast(Type[S3FilesStore], cls.STORE_SCHEMES["s3"])
s3store.AWS_ACCESS_KEY_ID = settings["AWS_ACCESS_KEY_ID"]
s3store.AWS_SECRET_ACCESS_KEY = settings["AWS_SECRET_ACCESS_KEY"]
s3store.AWS_SESSION_TOKEN = settings["AWS_SESSION_TOKEN"]
@ -107,11 +120,15 @@ class ImagesPipeline(FilesPipeline):
s3store.AWS_VERIFY = settings["AWS_VERIFY"]
s3store.POLICY = settings["IMAGES_STORE_S3_ACL"]
gcs_store = cls.STORE_SCHEMES["gs"]
gcs_store: Type[GCSFilesStore] = cast(
Type[GCSFilesStore], cls.STORE_SCHEMES["gs"]
)
gcs_store.GCS_PROJECT_ID = settings["GCS_PROJECT_ID"]
gcs_store.POLICY = settings["IMAGES_STORE_GCS_ACL"] or None
ftp_store = cls.STORE_SCHEMES["ftp"]
ftp_store: Type[FTPFilesStore] = cast(
Type[FTPFilesStore], cls.STORE_SCHEMES["ftp"]
)
ftp_store.FTP_USERNAME = settings["FTP_USER"]
ftp_store.FTP_PASSWORD = settings["FTP_PASSWORD"]
ftp_store.USE_ACTIVE_MODE = settings.getbool("FEED_STORAGE_FTP_ACTIVE")

View File

@ -1,6 +1,9 @@
from __future__ import annotations
import functools
import logging
from collections import defaultdict
from typing import TYPE_CHECKING
from twisted.internet.defer import Deferred, DeferredList
from twisted.python.failure import Failure
@ -12,6 +15,11 @@ from scrapy.utils.defer import defer_result, mustbe_deferred
from scrapy.utils.log import failure_to_exc_info
from scrapy.utils.misc import arg_to_iter
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
logger = logging.getLogger(__name__)
@ -67,9 +75,9 @@ class MediaPipeline:
return formatted_key
@classmethod
def from_crawler(cls, crawler):
def from_crawler(cls, crawler) -> Self:
try:
pipe = cls.from_settings(crawler.settings)
pipe = cls.from_settings(crawler.settings) # type: ignore[attr-defined]
except AttributeError:
pipe = cls()
pipe.crawler = crawler

View File

@ -5,8 +5,10 @@ for scraping typical web sites that requires crawling pages.
See documentation in docs/topics/spiders.rst
"""
from __future__ import annotations
import copy
from typing import AsyncIterable, Awaitable, Sequence
from typing import TYPE_CHECKING, AsyncIterable, Awaitable, Sequence
from scrapy.http import HtmlResponse, Request, Response
from scrapy.linkextractors import LinkExtractor
@ -14,6 +16,10 @@ from scrapy.spiders import Spider
from scrapy.utils.asyncgen import collect_asyncgen
from scrapy.utils.spider import iterate_spider_output
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
def _identity(x):
return x
@ -140,9 +146,9 @@ class CrawlSpider(Spider):
self._rules[-1]._compile(self)
@classmethod
def from_crawler(cls, crawler, *args, **kwargs):
def from_crawler(cls, crawler, *args, **kwargs) -> Self:
spider = super().from_crawler(crawler, *args, **kwargs)
spider._follow_links = crawler.settings.getbool(
spider._follow_links = crawler.settings.getbool( # type: ignore[attr-defined]
"CRAWLSPIDER_FOLLOW_LINKS", True
)
return spider

View File

@ -1,3 +1,5 @@
from __future__ import annotations
import logging
import re
from typing import TYPE_CHECKING, Any
@ -26,7 +28,7 @@ class SitemapSpider(Spider):
_warn_size: int
@classmethod
def from_crawler(cls, crawler: "Crawler", *args: Any, **kwargs: Any) -> "Self":
def from_crawler(cls, crawler: Crawler, *args: Any, **kwargs: Any) -> Self:
spider = super().from_crawler(crawler, *args, **kwargs)
spider._max_size = getattr(
spider, "download_maxsize", spider.settings.getint("DOWNLOAD_MAXSIZE")

View File

@ -3,6 +3,8 @@ This module provides some useful functions for working with
scrapy.http.Request objects
"""
from __future__ import annotations
import hashlib
import json
import warnings
@ -32,6 +34,9 @@ from scrapy.utils.misc import load_object
from scrapy.utils.python import to_bytes, to_unicode
if TYPE_CHECKING:
# typing.Self requires Python 3.11
from typing_extensions import Self
from scrapy.crawler import Crawler
@ -133,10 +138,10 @@ class RequestFingerprinter:
"""
@classmethod
def from_crawler(cls, crawler):
def from_crawler(cls, crawler) -> Self:
return cls(crawler)
def __init__(self, crawler: Optional["Crawler"] = None):
def __init__(self, crawler: Optional[Crawler] = None):
if crawler:
implementation = crawler.settings.get(
"REQUEST_FINGERPRINTER_IMPLEMENTATION"