From 74f062fe3d473297cb15a1a3221ba211f99157f8 Mon Sep 17 00:00:00 2001 From: Adrian Date: Tue, 11 Aug 2026 18:23:25 +0200 Subject: [PATCH 1/4] Upgrade twinecheck (#7979) --- tox.ini | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tox.ini b/tox.ini index 3069febbf..a6fc76d3e 100644 --- a/tox.ini +++ b/tox.ini @@ -124,7 +124,7 @@ commands = [testenv:twinecheck] basepython = python3 deps = - twine==6.2.0 + twine==7.0.0 build==1.5.0 commands = python -m build --sdist From 52cc2da72d5b12a184a1c89e6651352d78e3d2cb Mon Sep 17 00:00:00 2001 From: Adrian Date: Wed, 12 Aug 2026 08:23:41 +0200 Subject: [PATCH 2/4] Deprecate PeriodicLog.multiplier instead of removing it altogether (#7982) --- scrapy/extensions/periodic_log.py | 22 +++++++++++++++++++++- tests/test_extension_periodic_log.py | 16 +++++++++++++++- 2 files changed, 36 insertions(+), 2 deletions(-) diff --git a/scrapy/extensions/periodic_log.py b/scrapy/extensions/periodic_log.py index adffbcbc4..1b66eae4a 100644 --- a/scrapy/extensions/periodic_log.py +++ b/scrapy/extensions/periodic_log.py @@ -1,11 +1,12 @@ from __future__ import annotations import logging +import warnings from datetime import datetime, timezone from typing import TYPE_CHECKING, Any from scrapy import Spider, signals -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.utils.asyncio import AsyncioLoopingCall, create_looping_call from scrapy.utils.serialize import ScrapyJSONEncoder @@ -38,6 +39,7 @@ class PeriodicLog: ): self.stats: StatsCollector = stats self.interval: float = interval + self._multiplier: float = 60.0 / interval self.task: AsyncioLoopingCall | LoopingCall | None = None self.encoder: JSONEncoder = ScrapyJSONEncoder(sort_keys=True, indent=4) self.ext_stats_enabled: bool = bool(ext_stats) @@ -56,6 +58,24 @@ class PeriodicLog: ) self.ext_timing_enabled: bool = ext_timing_enabled + @property + def multiplier(self) -> float: + warnings.warn( + "The PeriodicLog.multiplier attribute is deprecated.", + ScrapyDeprecationWarning, + stacklevel=2, + ) + return self._multiplier + + @multiplier.setter + def multiplier(self, value: float) -> None: + warnings.warn( + "The PeriodicLog.multiplier attribute is deprecated.", + ScrapyDeprecationWarning, + stacklevel=2, + ) + self._multiplier = value + @classmethod def from_crawler(cls, crawler: Crawler) -> Self: interval: float = crawler.settings.getfloat("LOGSTATS_INTERVAL") diff --git a/tests/test_extension_periodic_log.py b/tests/test_extension_periodic_log.py index 340f8e28e..fd7e3958d 100644 --- a/tests/test_extension_periodic_log.py +++ b/tests/test_extension_periodic_log.py @@ -7,7 +7,7 @@ from typing import TYPE_CHECKING, Any import pytest -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.extensions.periodic_log import PeriodicLog from scrapy.utils.misc import build_from_crawler from scrapy.utils.test import get_crawler @@ -249,3 +249,17 @@ class TestPeriodicLog: assert data["time"]["log_interval_real"] >= 0 assert data["time"]["elapsed"] >= 0 assert data["time"]["start_time"] <= data["time"]["utcnow"] + + def test_multiplier_deprecated(self) -> None: + crawler = get_crawler( + MetaSpider, + {"PERIODIC_LOG_TIMING_ENABLED": True, "LOGSTATS_INTERVAL": 30}, + ) + crawler._apply_settings() + ext = build_from_crawler(PeriodicLog, crawler) + with pytest.warns(ScrapyDeprecationWarning): + assert ext.multiplier == 2.0 + with pytest.warns(ScrapyDeprecationWarning): + ext.multiplier = 3.0 + with pytest.warns(ScrapyDeprecationWarning): + assert ext.multiplier == 3.0 From 65b37286cc56b6a87c4d1eca6bdff79f658f09d2 Mon Sep 17 00:00:00 2001 From: Adrian Date: Wed, 12 Aug 2026 17:34:25 +0200 Subject: [PATCH 3/4] Document the ITEM_PROCESSOR setting (#7983) --- docs/topics/settings.rst | 14 ++++++++++++++ scrapy/pipelines/__init__.py | 19 ++++++++++++++++++- 2 files changed, 32 insertions(+), 1 deletion(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 27ef3f7ef..010c9d1c6 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1529,6 +1529,20 @@ Default: ``{}`` A dict containing the pipelines enabled by default in Scrapy. You should never modify this setting in your project, modify :setting:`ITEM_PIPELINES` instead. +.. setting:: ITEM_PROCESSOR + +ITEM_PROCESSOR +-------------- + +Default: ``"scrapy.pipelines.ItemPipelineManager"`` + +The :ref:`component ` that builds the :ref:`item pipeline +` from :setting:`ITEM_PIPELINES` and runs scraped items +through it. It must implement :class:`~scrapy.pipelines.ItemProcessorProtocol`. + +.. autoclass:: scrapy.pipelines.ItemProcessorProtocol + :members: + .. setting:: JOBDIR diff --git a/scrapy/pipelines/__init__.py b/scrapy/pipelines/__init__.py index 383c461e6..e6483d939 100644 --- a/scrapy/pipelines/__init__.py +++ b/scrapy/pipelines/__init__.py @@ -8,7 +8,7 @@ from __future__ import annotations import asyncio import warnings -from typing import TYPE_CHECKING, Any, cast +from typing import TYPE_CHECKING, Any, Protocol, cast from twisted.internet.defer import Deferred, DeferredList, FirstError @@ -28,6 +28,23 @@ if TYPE_CHECKING: from scrapy.settings import Settings +class ItemProcessorProtocol(Protocol): + """Protocol for item processor implementations. + + See :setting:`ITEM_PROCESSOR`. + """ + + async def open_spider_async(self) -> None: + """Get the item processor ready to process items.""" + + async def process_item_async(self, item: Any) -> Any: + """Return the processed *item*, or raise + :exc:`~scrapy.exceptions.DropItem` to drop it.""" + + async def close_spider_async(self) -> None: + """Release any resource that the item processor is using.""" + + class ItemPipelineManager(MiddlewareManager): component_name = "item pipeline" From 8bb06bf00b4a098f9c8349743df2bf1539a3a3d3 Mon Sep 17 00:00:00 2001 From: Adrian Date: Wed, 12 Aug 2026 20:33:38 +0200 Subject: [PATCH 4/4] Stop marking the Twisted-based HTTP/2 download handler as experimental, and privatize its API (#7986) --- conftest.py | 2 +- docs/topics/download-handlers.rst | 6 ------ scrapy/core/{http2 => _http2}/__init__.py | 0 scrapy/core/{http2 => _http2}/agent.py | 2 +- scrapy/core/{http2 => _http2}/protocol.py | 2 +- scrapy/core/{http2 => _http2}/stream.py | 2 +- scrapy/core/downloader/handlers/http2.py | 2 +- tests/test_downloader_handler_twisted_http2.py | 2 +- tests/test_http2_client_protocol.py | 14 +++++++------- 9 files changed, 13 insertions(+), 19 deletions(-) rename scrapy/core/{http2 => _http2}/__init__.py (100%) rename scrapy/core/{http2 => _http2}/agent.py (98%) rename scrapy/core/{http2 => _http2}/protocol.py (99%) rename scrapy/core/{http2 => _http2}/stream.py (99%) diff --git a/conftest.py b/conftest.py index 5a535c168..7a5f5b85a 100644 --- a/conftest.py +++ b/conftest.py @@ -47,7 +47,7 @@ if not H2_ENABLED: collect_ignore.extend( ( "scrapy/core/downloader/handlers/http2.py", - *_py_files("scrapy/core/http2"), + *_py_files("scrapy/core/_http2"), ) ) diff --git a/docs/topics/download-handlers.rst b/docs/topics/download-handlers.rst index 433a6d139..d95e7b52b 100644 --- a/docs/topics/download-handlers.rst +++ b/docs/topics/download-handlers.rst @@ -187,12 +187,6 @@ If you want to use this handler you need to replace the default one for the Features and limitations ^^^^^^^^^^^^^^^^^^^^^^^^ -.. warning:: - - This handler is experimental, and not yet recommended for production - environments. Future Scrapy versions may introduce related changes without - a deprecation period or warning. - =========================== ================================================ HTTP proxies No (not implemented) SOCKS proxies No (not supported by the library) diff --git a/scrapy/core/http2/__init__.py b/scrapy/core/_http2/__init__.py similarity index 100% rename from scrapy/core/http2/__init__.py rename to scrapy/core/_http2/__init__.py diff --git a/scrapy/core/http2/agent.py b/scrapy/core/_http2/agent.py similarity index 98% rename from scrapy/core/http2/agent.py rename to scrapy/core/_http2/agent.py index 042557208..38be86b22 100644 --- a/scrapy/core/http2/agent.py +++ b/scrapy/core/_http2/agent.py @@ -14,8 +14,8 @@ from twisted.web.client import ( ) from twisted.web.error import SchemeNotSupported +from scrapy.core._http2.protocol import H2ClientFactory, H2ClientProtocol from scrapy.core.downloader.contextfactory import _AcceptableProtocolsContextFactory -from scrapy.core.http2.protocol import H2ClientFactory, H2ClientProtocol if TYPE_CHECKING: from twisted.internet.base import ReactorBase diff --git a/scrapy/core/http2/protocol.py b/scrapy/core/_http2/protocol.py similarity index 99% rename from scrapy/core/http2/protocol.py rename to scrapy/core/_http2/protocol.py index 2d59aba31..aff985d94 100644 --- a/scrapy/core/http2/protocol.py +++ b/scrapy/core/_http2/protocol.py @@ -31,7 +31,7 @@ from twisted.internet.ssl import Certificate from twisted.protocols.policies import TimeoutMixin from zope.interface import implementer -from scrapy.core.http2.stream import Stream, StreamCloseReason +from scrapy.core._http2.stream import Stream, StreamCloseReason from scrapy.exceptions import DownloadTimeoutError from scrapy.http import Request, Response from scrapy.utils.deprecate import warn_on_deprecated_spider_attribute diff --git a/scrapy/core/http2/stream.py b/scrapy/core/_http2/stream.py similarity index 99% rename from scrapy/core/http2/stream.py rename to scrapy/core/_http2/stream.py index 4fc300d90..4a07d198b 100644 --- a/scrapy/core/http2/stream.py +++ b/scrapy/core/_http2/stream.py @@ -27,7 +27,7 @@ from scrapy.utils.httpobj import urlparse_cached if TYPE_CHECKING: from collections.abc import Sequence - from scrapy.core.http2.protocol import H2ClientProtocol + from scrapy.core._http2.protocol import H2ClientProtocol from scrapy.crawler import Crawler from scrapy.http import Request, Response diff --git a/scrapy/core/downloader/handlers/http2.py b/scrapy/core/downloader/handlers/http2.py index 9b3d4fbd4..935a2ae4c 100644 --- a/scrapy/core/downloader/handlers/http2.py +++ b/scrapy/core/downloader/handlers/http2.py @@ -4,9 +4,9 @@ from time import monotonic from typing import TYPE_CHECKING from urllib.parse import urldefrag +from scrapy.core._http2.agent import H2Agent, H2ConnectionPool from scrapy.core.downloader.contextfactory import _load_context_factory_from_settings from scrapy.core.downloader.handlers.base import BaseDownloadHandler -from scrapy.core.http2.agent import H2Agent, H2ConnectionPool from scrapy.exceptions import ( DownloadTimeoutError, NotConfigured, diff --git a/tests/test_downloader_handler_twisted_http2.py b/tests/test_downloader_handler_twisted_http2.py index 2c3954b5e..95e1ff633 100644 --- a/tests/test_downloader_handler_twisted_http2.py +++ b/tests/test_downloader_handler_twisted_http2.py @@ -143,7 +143,7 @@ class TestHttp2(H2DownloadHandlerMixin, TestHttpsBase): response = await download_handler.download_request(request) assert response.text == actual_content_length assert ( - "scrapy.core.http2.stream", + "scrapy.core._http2.stream", logging.WARNING, f"Ignoring bad Content-Length header " f"{bad_content_length!r} of request {request}, sending " diff --git a/tests/test_http2_client_protocol.py b/tests/test_http2_client_protocol.py index 431b7a458..161015231 100644 --- a/tests/test_http2_client_protocol.py +++ b/tests/test_http2_client_protocol.py @@ -36,7 +36,7 @@ from tests.mockserver.utils import ssl_context_factory if TYPE_CHECKING: from collections.abc import AsyncGenerator, Callable, Coroutine, Generator - from scrapy.core.http2.protocol import H2ClientProtocol + from scrapy.core._http2.protocol import H2ClientProtocol pytestmark = [ @@ -242,7 +242,7 @@ class TestHttps2ClientProtocol: ) -> AsyncGenerator[H2ClientProtocol]: from twisted.internet import reactor - from scrapy.core.http2.protocol import H2ClientFactory # noqa: PLC0415 + from scrapy.core._http2.protocol import H2ClientFactory # noqa: PLC0415 client_options = optionsForClientTLS( hostname=self.host, @@ -460,7 +460,7 @@ class TestHttps2ClientProtocol: def test_invalid_negotiated_protocol( self, server_port: int, client: H2ClientProtocol ) -> Generator[Deferred[Any], Any, None]: - with mock.patch("scrapy.core.http2.protocol.PROTOCOL_NAME", new=b"not-h2"): + with mock.patch("scrapy.core._http2.protocol.PROTOCOL_NAME", new=b"not-h2"): request = Request(url=self.get_url(server_port, "/status?n=200")) with pytest.raises(ResponseFailed): yield make_request_dfd(client, request) @@ -527,7 +527,7 @@ class TestHttps2ClientProtocol: expected_body: bytes, caplog: pytest.LogCaptureFixture, ) -> None: - with caplog.at_level("WARNING", "scrapy.core.http2.stream"): + with caplog.at_level("WARNING", "scrapy.core._http2.stream"): response = await make_request(client, request) assert response.status == 200 assert response.body == expected_body @@ -605,7 +605,7 @@ class TestHttps2ClientProtocol: def assert_inactive_stream(failure): assert failure.check(ResponseFailed) is not None - from scrapy.core.http2.stream import InactiveStreamClosed # noqa: PLC0415 + from scrapy.core._http2.stream import InactiveStreamClosed # noqa: PLC0415 assert any( isinstance(e, InactiveStreamClosed) for e in failure.value.reasons @@ -692,7 +692,7 @@ class TestHttps2ClientProtocol: @staticmethod async def _check_invalid_netloc(client: H2ClientProtocol, url: str) -> None: - from scrapy.core.http2.stream import InvalidHostname # noqa: PLC0415 + from scrapy.core._http2.stream import InvalidHostname # noqa: PLC0415 request = Request(url) with pytest.raises(InvalidHostname) as exc_info: @@ -737,7 +737,7 @@ class TestHttps2ClientProtocol: yield make_request_dfd(client, request) for err in exc_info.value.reasons: - from scrapy.core.http2.protocol import H2ClientProtocol # noqa: PLC0415 + from scrapy.core._http2.protocol import H2ClientProtocol # noqa: PLC0415 if isinstance(err, DownloadTimeoutError): assert (