diff --git a/conftest.py b/conftest.py index 5a535c168..ad11cedcc 100644 --- a/conftest.py +++ b/conftest.py @@ -47,7 +47,7 @@ if not H2_ENABLED: collect_ignore.extend( ( "scrapy/core/downloader/handlers/http2.py", - *_py_files("scrapy/core/http2"), + *_py_files("scrapy/core/_http2"), ) ) @@ -107,6 +107,14 @@ def pytest_configure(config): install_reactor_import_hook() +def pytest_collection_modifyitems(items): + for item in items: + if item.get_closest_marker("requires_internet"): + # Requests to real websites fail every now and then in CI for + # reasons unrelated to the code under test. + item.add_marker(pytest.mark.flaky(reruns=2, reruns_delay=5)) + + def pytest_runtest_setup(item): # Skip tests based on reactor markers reactor = item.config.getoption("--reactor") diff --git a/docs/topics/download-handlers.rst b/docs/topics/download-handlers.rst index 433a6d139..2d04b621c 100644 --- a/docs/topics/download-handlers.rst +++ b/docs/topics/download-handlers.rst @@ -187,12 +187,6 @@ If you want to use this handler you need to replace the default one for the Features and limitations ^^^^^^^^^^^^^^^^^^^^^^^^ -.. warning:: - - This handler is experimental, and not yet recommended for production - environments. Future Scrapy versions may introduce related changes without - a deprecation period or warning. - =========================== ================================================ HTTP proxies No (not implemented) SOCKS proxies No (not supported by the library) @@ -215,12 +209,8 @@ Known limitations of the HTTP/2 support: - No support for HTTP/2 Cleartext (h2c), since no major browser supports HTTP/2 unencrypted (refer `http2 faq`_). -- No setting to specify a maximum `frame size`_ larger than the default - value, 16384. Connections to servers that send a larger frame will fail. - - No support for `server pushes`_, which are ignored. -.. _frame size: https://datatracker.ietf.org/doc/html/rfc7540#section-4.2 .. _http2 faq: https://http2.github.io/faq/#does-http2-require-encryption .. _server pushes: https://datatracker.ietf.org/doc/html/rfc7540#section-8.2 diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index fd793442e..a315e0a6e 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1549,6 +1549,28 @@ to set this on a per-request basis. It supports the same values as the setting, and it takes precedence over the setting, e.g. ``False`` makes a request handle no status code even when the setting is ``True``. +.. setting:: HTTP2_MAX_FRAME_SIZE + +HTTP2_MAX_FRAME_SIZE +-------------------- + +.. versionadded:: VERSION + +Default: ``16384`` + +Maximum `frame size`_, in bytes, that servers may send, between ``16384`` and +``16777215``. Connections to servers that send a larger frame fail. + +Raise it for servers that send larger frames regardless of this value. Note +that :setting:`DOWNLOAD_MAXSIZE` and :setting:`DOWNLOAD_WARNSIZE` are checked +once per received frame, so a higher value allows a response to exceed them by +more before being caught. + +:class:`~scrapy.core.downloader.handlers._httpx.HttpxDownloadHandler` ignores +this setting, as ``httpx`` does not allow configuring the frame size. + +.. _frame size: https://datatracker.ietf.org/doc/html/rfc7540#section-4.2 + .. setting:: ITEM_PIPELINES ITEM_PIPELINES @@ -1579,6 +1601,20 @@ Default: ``{}`` A dict containing the pipelines enabled by default in Scrapy. You should never modify this setting in your project, modify :setting:`ITEM_PIPELINES` instead. +.. setting:: ITEM_PROCESSOR + +ITEM_PROCESSOR +-------------- + +Default: ``"scrapy.pipelines.ItemPipelineManager"`` + +The :ref:`component ` that builds the :ref:`item pipeline +` from :setting:`ITEM_PIPELINES` and runs scraped items +through it. It must implement :class:`~scrapy.pipelines.ItemProcessorProtocol`. + +.. autoclass:: scrapy.pipelines.ItemProcessorProtocol + :members: + .. setting:: JOBDIR diff --git a/scrapy/core/http2/__init__.py b/scrapy/core/_http2/__init__.py similarity index 100% rename from scrapy/core/http2/__init__.py rename to scrapy/core/_http2/__init__.py diff --git a/scrapy/core/http2/agent.py b/scrapy/core/_http2/agent.py similarity index 98% rename from scrapy/core/http2/agent.py rename to scrapy/core/_http2/agent.py index 042557208..38be86b22 100644 --- a/scrapy/core/http2/agent.py +++ b/scrapy/core/_http2/agent.py @@ -14,8 +14,8 @@ from twisted.web.client import ( ) from twisted.web.error import SchemeNotSupported +from scrapy.core._http2.protocol import H2ClientFactory, H2ClientProtocol from scrapy.core.downloader.contextfactory import _AcceptableProtocolsContextFactory -from scrapy.core.http2.protocol import H2ClientFactory, H2ClientProtocol if TYPE_CHECKING: from twisted.internet.base import ReactorBase diff --git a/scrapy/core/http2/protocol.py b/scrapy/core/_http2/protocol.py similarity index 98% rename from scrapy/core/http2/protocol.py rename to scrapy/core/_http2/protocol.py index 2d59aba31..1e673ce8b 100644 --- a/scrapy/core/http2/protocol.py +++ b/scrapy/core/_http2/protocol.py @@ -21,6 +21,7 @@ from h2.events import ( WindowUpdated, ) from h2.exceptions import FrameTooLargeError, H2Error +from h2.settings import SettingCodes from twisted.internet.interfaces import ( IAddress, IHandshakeListener, @@ -31,7 +32,7 @@ from twisted.internet.ssl import Certificate from twisted.protocols.policies import TimeoutMixin from zope.interface import implementer -from scrapy.core.http2.stream import Stream, StreamCloseReason +from scrapy.core._http2.stream import Stream, StreamCloseReason from scrapy.exceptions import DownloadTimeoutError from scrapy.http import Request, Response from scrapy.utils.deprecate import warn_on_deprecated_spider_attribute @@ -261,6 +262,9 @@ class H2ClientProtocol(Protocol, TimeoutMixin): # Initiate H2 Connection self.conn.initiate_connection() + max_frame_size = self._crawler.settings.getint("HTTP2_MAX_FRAME_SIZE") + if max_frame_size != self.conn.local_settings.max_frame_size: + self.conn.update_settings({SettingCodes.MAX_FRAME_SIZE: max_frame_size}) self._write_to_transport() def _lose_connection_with_error(self, errors: list[BaseException]) -> None: diff --git a/scrapy/core/http2/stream.py b/scrapy/core/_http2/stream.py similarity index 99% rename from scrapy/core/http2/stream.py rename to scrapy/core/_http2/stream.py index 4fc300d90..4a07d198b 100644 --- a/scrapy/core/http2/stream.py +++ b/scrapy/core/_http2/stream.py @@ -27,7 +27,7 @@ from scrapy.utils.httpobj import urlparse_cached if TYPE_CHECKING: from collections.abc import Sequence - from scrapy.core.http2.protocol import H2ClientProtocol + from scrapy.core._http2.protocol import H2ClientProtocol from scrapy.crawler import Crawler from scrapy.http import Request, Response diff --git a/scrapy/core/downloader/handlers/http2.py b/scrapy/core/downloader/handlers/http2.py index 9b3d4fbd4..935a2ae4c 100644 --- a/scrapy/core/downloader/handlers/http2.py +++ b/scrapy/core/downloader/handlers/http2.py @@ -4,9 +4,9 @@ from time import monotonic from typing import TYPE_CHECKING from urllib.parse import urldefrag +from scrapy.core._http2.agent import H2Agent, H2ConnectionPool from scrapy.core.downloader.contextfactory import _load_context_factory_from_settings from scrapy.core.downloader.handlers.base import BaseDownloadHandler -from scrapy.core.http2.agent import H2Agent, H2ConnectionPool from scrapy.exceptions import ( DownloadTimeoutError, NotConfigured, diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index fbc6f2530..5d1ec246c 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -9,7 +9,7 @@ from __future__ import annotations import logging from collections.abc import AsyncIterator, Callable, Coroutine, Iterable from functools import wraps -from inspect import isasyncgenfunction +from inspect import isasyncgenfunction, iscoroutine from itertools import islice from typing import TYPE_CHECKING, Any, TypeAlias, TypeVar from warnings import warn @@ -244,8 +244,20 @@ class SpiderMiddlewareManager(MiddlewareManager): warn(msg, category=ScrapyDeprecationWarning, stacklevel=2) self._set_compat_spider(spider) start = self._spider.start() + if not hasattr(start, "__aiter__"): + if iscoroutine(start): + start.close() + start = self._reject_start(start) return await self._process_chain("process_start", start) + async def _reject_start(self, start: Any) -> AsyncIterator[Any]: + raise TypeError( + f"{global_object_name(type(self._spider))}.start() must be an" + f" asynchronous generator, i.e. an async def method with yield" + f" statements, got {type(start)}" + ) + yield # pylint: disable=unreachable # makes this method an asynchronous generator + # This method is only needed until _async compatibility methods are removed. @staticmethod def _get_process_spider_output(mw: Any) -> Callable[..., Any] | None: diff --git a/scrapy/extensions/periodic_log.py b/scrapy/extensions/periodic_log.py index adffbcbc4..1b66eae4a 100644 --- a/scrapy/extensions/periodic_log.py +++ b/scrapy/extensions/periodic_log.py @@ -1,11 +1,12 @@ from __future__ import annotations import logging +import warnings from datetime import datetime, timezone from typing import TYPE_CHECKING, Any from scrapy import Spider, signals -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.utils.asyncio import AsyncioLoopingCall, create_looping_call from scrapy.utils.serialize import ScrapyJSONEncoder @@ -38,6 +39,7 @@ class PeriodicLog: ): self.stats: StatsCollector = stats self.interval: float = interval + self._multiplier: float = 60.0 / interval self.task: AsyncioLoopingCall | LoopingCall | None = None self.encoder: JSONEncoder = ScrapyJSONEncoder(sort_keys=True, indent=4) self.ext_stats_enabled: bool = bool(ext_stats) @@ -56,6 +58,24 @@ class PeriodicLog: ) self.ext_timing_enabled: bool = ext_timing_enabled + @property + def multiplier(self) -> float: + warnings.warn( + "The PeriodicLog.multiplier attribute is deprecated.", + ScrapyDeprecationWarning, + stacklevel=2, + ) + return self._multiplier + + @multiplier.setter + def multiplier(self, value: float) -> None: + warnings.warn( + "The PeriodicLog.multiplier attribute is deprecated.", + ScrapyDeprecationWarning, + stacklevel=2, + ) + self._multiplier = value + @classmethod def from_crawler(cls, crawler: Crawler) -> Self: interval: float = crawler.settings.getfloat("LOGSTATS_INTERVAL") diff --git a/scrapy/pipelines/__init__.py b/scrapy/pipelines/__init__.py index 383c461e6..e6483d939 100644 --- a/scrapy/pipelines/__init__.py +++ b/scrapy/pipelines/__init__.py @@ -8,7 +8,7 @@ from __future__ import annotations import asyncio import warnings -from typing import TYPE_CHECKING, Any, cast +from typing import TYPE_CHECKING, Any, Protocol, cast from twisted.internet.defer import Deferred, DeferredList, FirstError @@ -28,6 +28,23 @@ if TYPE_CHECKING: from scrapy.settings import Settings +class ItemProcessorProtocol(Protocol): + """Protocol for item processor implementations. + + See :setting:`ITEM_PROCESSOR`. + """ + + async def open_spider_async(self) -> None: + """Get the item processor ready to process items.""" + + async def process_item_async(self, item: Any) -> Any: + """Return the processed *item*, or raise + :exc:`~scrapy.exceptions.DropItem` to drop it.""" + + async def close_spider_async(self) -> None: + """Release any resource that the item processor is using.""" + + class ItemPipelineManager(MiddlewareManager): component_name = "item pipeline" diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index 1a96f8f50..74ce27a8f 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -109,6 +109,7 @@ __all__ = [ "FTP_USER", "GCS_PROJECT_ID", "HANDLE_HTTP_CODES", + "HTTP2_MAX_FRAME_SIZE", "HTTPAUTH_DOMAIN", "HTTPAUTH_PASS", "HTTPAUTH_USER", @@ -405,6 +406,8 @@ GCS_PROJECT_ID = None HANDLE_HTTP_CODES = None +HTTP2_MAX_FRAME_SIZE = 16384 + HTTPAUTH_USER = "" HTTPAUTH_PASS = "" HTTPAUTH_DOMAIN = None diff --git a/tests/test_downloader_handler_twisted_http2.py b/tests/test_downloader_handler_twisted_http2.py index 2c3954b5e..95e1ff633 100644 --- a/tests/test_downloader_handler_twisted_http2.py +++ b/tests/test_downloader_handler_twisted_http2.py @@ -143,7 +143,7 @@ class TestHttp2(H2DownloadHandlerMixin, TestHttpsBase): response = await download_handler.download_request(request) assert response.text == actual_content_length assert ( - "scrapy.core.http2.stream", + "scrapy.core._http2.stream", logging.WARNING, f"Ignoring bad Content-Length header " f"{bad_content_length!r} of request {request}, sending " diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 50ff5e63f..6eabe91ad 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -112,7 +112,7 @@ class StorageTestMixin(TestBase): raise NotImplementedError def test_storage(self): - with self._storage(HTTPCACHE_EXPIRATION_SECS=1) as (storage, crawler): + with self._storage(HTTPCACHE_EXPIRATION_SECS=100) as (storage, crawler): request2 = self.request.copy() assert storage.retrieve_response(crawler.spider, request2) is None diff --git a/tests/test_engine_loop.py b/tests/test_engine_loop.py index 1ecf8b8de..8e5197df7 100644 --- a/tests/test_engine_loop.py +++ b/tests/test_engine_loop.py @@ -4,6 +4,8 @@ from collections import deque from logging import ERROR from typing import TYPE_CHECKING, Any +import pytest + from scrapy import Request, Spider, signals from scrapy.core.scheduler import BaseScheduler from scrapy.exceptions import CloseSpider @@ -13,7 +15,7 @@ from tests.mockserver.http import MockServer from tests.utils.decorators import coroutine_test if TYPE_CHECKING: - import pytest + from collections.abc import Iterator from scrapy.http import Response @@ -50,6 +52,27 @@ class MemoryScheduler(BaseScheduler): self.paused = False +class NoneStartSpider(Spider): + name = "test" + + def start(self) -> None: # type: ignore[override] + return None + + +class CoroutineStartSpider(Spider): + name = "test" + + async def start(self) -> None: # type: ignore[override] + return None + + +class SyncStartSpider(Spider): + name = "test" + + def start(self) -> Iterator[Request]: # type: ignore[override] + yield Request("data:,a") + + class TestMain: @coroutine_test async def test_sleep(self): @@ -141,6 +164,34 @@ class TestMain: assert crawler.stats.get_value("finish_reason") == "shutdown" assert not actual_urls + @pytest.mark.parametrize( + ("spider_cls", "expected_type"), + [ + (NoneStartSpider, ""), + (CoroutineStartSpider, ""), + (SyncStartSpider, ""), + ], + ) + @coroutine_test + async def test_start_not_an_async_generator( + self, + spider_cls: type[Spider], + expected_type: str, + caplog: pytest.LogCaptureFixture, + ) -> None: + crawler = get_crawler(spider_cls) + + caplog.clear() + with caplog.at_level(ERROR): + await crawler.crawl_async() + + assert ( + f"{spider_cls.__name__}.start() must be an asynchronous generator," + f" i.e. an async def method with yield statements, got {expected_type}" + ) in caplog.text + assert crawler.stats + assert crawler.stats.get_value("finish_reason") == "start_error" + @coroutine_test async def test_start_error(self, caplog: pytest.LogCaptureFixture) -> None: class TestSpider(Spider): diff --git a/tests/test_extension_periodic_log.py b/tests/test_extension_periodic_log.py index 340f8e28e..fd7e3958d 100644 --- a/tests/test_extension_periodic_log.py +++ b/tests/test_extension_periodic_log.py @@ -7,7 +7,7 @@ from typing import TYPE_CHECKING, Any import pytest -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.extensions.periodic_log import PeriodicLog from scrapy.utils.misc import build_from_crawler from scrapy.utils.test import get_crawler @@ -249,3 +249,17 @@ class TestPeriodicLog: assert data["time"]["log_interval_real"] >= 0 assert data["time"]["elapsed"] >= 0 assert data["time"]["start_time"] <= data["time"]["utcnow"] + + def test_multiplier_deprecated(self) -> None: + crawler = get_crawler( + MetaSpider, + {"PERIODIC_LOG_TIMING_ENABLED": True, "LOGSTATS_INTERVAL": 30}, + ) + crawler._apply_settings() + ext = build_from_crawler(PeriodicLog, crawler) + with pytest.warns(ScrapyDeprecationWarning): + assert ext.multiplier == 2.0 + with pytest.warns(ScrapyDeprecationWarning): + ext.multiplier = 3.0 + with pytest.warns(ScrapyDeprecationWarning): + assert ext.multiplier == 3.0 diff --git a/tests/test_http2_client_protocol.py b/tests/test_http2_client_protocol.py index 431b7a458..e96a5484e 100644 --- a/tests/test_http2_client_protocol.py +++ b/tests/test_http2_client_protocol.py @@ -36,7 +36,8 @@ from tests.mockserver.utils import ssl_context_factory if TYPE_CHECKING: from collections.abc import AsyncGenerator, Callable, Coroutine, Generator - from scrapy.core.http2.protocol import H2ClientProtocol + from scrapy.core._http2.protocol import H2ClientProtocol + from scrapy.crawler import Crawler pytestmark = [ @@ -75,7 +76,7 @@ class Data: STR_LARGE = generate_random_string(LARGE_SIZE) EXTRA_SMALL = generate_random_string(1024 * 15) - EXTRA_LARGE = generate_random_string((1024**2) * 15) + EXTRA_LARGE = generate_random_string(LARGE_SIZE) HTML_SMALL = make_html_body(STR_SMALL) HTML_LARGE = make_html_body(STR_LARGE) @@ -236,13 +237,20 @@ class TestHttps2ClientProtocol: ) + self.certificate_file.read_text(encoding="utf-8") return PrivateCertificate.loadPEM(pem) # type: ignore[no-any-return] + @pytest.fixture + def crawler(self, request: pytest.FixtureRequest) -> Crawler: + return get_crawler(settings_dict=getattr(request, "param", None)) + @async_yield_fixture # type: ignore[untyped-decorator] async def client( - self, server_port: int, client_certificate: PrivateCertificate + self, + server_port: int, + client_certificate: PrivateCertificate, + crawler: Crawler, ) -> AsyncGenerator[H2ClientProtocol]: from twisted.internet import reactor - from scrapy.core.http2.protocol import H2ClientFactory # noqa: PLC0415 + from scrapy.core._http2.protocol import H2ClientFactory # noqa: PLC0415 client_options = optionsForClientTLS( hostname=self.host, @@ -250,7 +258,7 @@ class TestHttps2ClientProtocol: acceptableProtocols=[b"h2"], ) uri = URI.fromBytes(bytes(self.get_url(server_port, "/"), "utf-8")) - h2_client_factory = H2ClientFactory(uri, get_crawler(), Deferred()) + h2_client_factory = H2ClientFactory(uri, crawler, Deferred()) client_endpoint = SSL4ClientEndpoint( reactor, self.host, server_port, client_options ) @@ -312,6 +320,17 @@ class TestHttps2ClientProtocol: request = Request(self.get_url(server_port, "/get-data-html-large")) await self._check_GET(client, request, Data.HTML_LARGE, 200) + @pytest.mark.parametrize( + "crawler", [{"HTTP2_MAX_FRAME_SIZE": 1024**2}], indirect=True + ) + @deferred_f_from_coro_f + async def test_GET_large_frames( + self, server_port: int, client: H2ClientProtocol + ) -> None: + request = Request(self.get_url(server_port, "/get-data-html-large")) + await self._check_GET(client, request, Data.HTML_LARGE, 200) + assert client.conn.local_settings.max_frame_size == 1024**2 + async def _check_GET_x10( self, client: H2ClientProtocol, @@ -460,7 +479,7 @@ class TestHttps2ClientProtocol: def test_invalid_negotiated_protocol( self, server_port: int, client: H2ClientProtocol ) -> Generator[Deferred[Any], Any, None]: - with mock.patch("scrapy.core.http2.protocol.PROTOCOL_NAME", new=b"not-h2"): + with mock.patch("scrapy.core._http2.protocol.PROTOCOL_NAME", new=b"not-h2"): request = Request(url=self.get_url(server_port, "/status?n=200")) with pytest.raises(ResponseFailed): yield make_request_dfd(client, request) @@ -527,7 +546,7 @@ class TestHttps2ClientProtocol: expected_body: bytes, caplog: pytest.LogCaptureFixture, ) -> None: - with caplog.at_level("WARNING", "scrapy.core.http2.stream"): + with caplog.at_level("WARNING", "scrapy.core._http2.stream"): response = await make_request(client, request) assert response.status == 200 assert response.body == expected_body @@ -605,7 +624,7 @@ class TestHttps2ClientProtocol: def assert_inactive_stream(failure): assert failure.check(ResponseFailed) is not None - from scrapy.core.http2.stream import InactiveStreamClosed # noqa: PLC0415 + from scrapy.core._http2.stream import InactiveStreamClosed # noqa: PLC0415 assert any( isinstance(e, InactiveStreamClosed) for e in failure.value.reasons @@ -692,7 +711,7 @@ class TestHttps2ClientProtocol: @staticmethod async def _check_invalid_netloc(client: H2ClientProtocol, url: str) -> None: - from scrapy.core.http2.stream import InvalidHostname # noqa: PLC0415 + from scrapy.core._http2.stream import InvalidHostname # noqa: PLC0415 request = Request(url) with pytest.raises(InvalidHostname) as exc_info: @@ -737,7 +756,7 @@ class TestHttps2ClientProtocol: yield make_request_dfd(client, request) for err in exc_info.value.reasons: - from scrapy.core.http2.protocol import H2ClientProtocol # noqa: PLC0415 + from scrapy.core._http2.protocol import H2ClientProtocol # noqa: PLC0415 if isinstance(err, DownloadTimeoutError): assert ( diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 608b2bbd9..3638c52b7 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -184,32 +184,31 @@ def test_inject_base_url(body: bytes) -> None: assert open_in_browser(resp, _openfunc=check_base_url) -def test_open_in_browser_redos_comment(): - MAX_CPU_TIME = 0.02 +def _assert_open_in_browser_is_fast(body: bytes) -> None: + # The exploit inputs are large enough that a vulnerable implementation + # needs seconds to go through them, while a safe one stays in the low + # milliseconds even on a slow interpreter. + max_cpu_time = 0.2 + response = HtmlResponse("https://example.com", body=body) + start_time = process_time() + open_in_browser(response, lambda url: True) + end_time = process_time() + assert end_time - start_time < max_cpu_time + + +def test_open_in_browser_redos_comment(): # Exploit input from # https://makenowjust-labs.github.io/recheck/playground/ # for // (old pattern to remove comments). - body = b"->" - response = HtmlResponse("https://example.com", body=body) - start_time = process_time() - open_in_browser(response, lambda url: True) - end_time = process_time() - assert end_time - start_time < MAX_CPU_TIME + _assert_open_in_browser_is_fast(b"->") def test_open_in_browser_redos_head(): - MAX_CPU_TIME = 0.02 - # Exploit input from # https://makenowjust-labs.github.io/recheck/playground/ # for /(|\s.*?>))/ (old pattern to find the head element). - body = b" AsyncGenerator[DownloadHandlerProtocol]: - crawler = get_crawler(DefaultSpider, settings_dict) + crawler = get_crawler( + DefaultSpider, {**REAL_WEBSITE_SETTINGS, **(settings_dict or {})} + ) crawler.spider = crawler._create_spider() dh = build_from_crawler(self.download_handler_cls, crawler) try: @@ -1555,7 +1563,9 @@ class TestRealWebsiteBase(ABC): @coroutine_test async def test_download_with_spider(self) -> None: - crawler = get_crawler(SingleRequestSpider, self.settings_dict) + crawler = get_crawler( + SingleRequestSpider, {**REAL_WEBSITE_SETTINGS, **(self.settings_dict or {})} + ) await maybe_deferred_to_future( crawler.crawl(seed=Request("https://books.toscrape.com/")) ) diff --git a/tox.ini b/tox.ini index 3069febbf..7ea1c57d7 100644 --- a/tox.ini +++ b/tox.ini @@ -44,6 +44,7 @@ deps = pygments pytest pytest-cov >= 7.0.0 + pytest-rerunfailures pytest-timeout pytest-xdist sybil >= 1.3.0 # https://github.com/cjw296/sybil/issues/20#issuecomment-605433422 @@ -124,7 +125,7 @@ commands = [testenv:twinecheck] basepython = python3 deps = - twine==6.2.0 + twine==7.0.0 build==1.5.0 commands = python -m build --sdist