Merge remote-tracking branch 'origin/master' into max-header-length

This commit is contained in:
Adrian Chaves 2026-08-12 22:04:06 +02:00
commit 72ab0cdc92
15 changed files with 103 additions and 25 deletions

View File

@ -47,7 +47,7 @@ if not H2_ENABLED:
collect_ignore.extend(
(
"scrapy/core/downloader/handlers/http2.py",
*_py_files("scrapy/core/http2"),
*_py_files("scrapy/core/_http2"),
)
)
@ -107,6 +107,14 @@ def pytest_configure(config):
install_reactor_import_hook()
def pytest_collection_modifyitems(items):
for item in items:
if item.get_closest_marker("requires_internet"):
# Requests to real websites fail every now and then in CI for
# reasons unrelated to the code under test.
item.add_marker(pytest.mark.flaky(reruns=2, reruns_delay=5))
def pytest_runtest_setup(item):
# Skip tests based on reactor markers
reactor = item.config.getoption("--reactor")

View File

@ -221,12 +221,6 @@ If you want to use this handler you need to replace the default one for the
Features and limitations
^^^^^^^^^^^^^^^^^^^^^^^^
.. warning::
This handler is experimental, and not yet recommended for production
environments. Future Scrapy versions may introduce related changes without
a deprecation period or warning.
=========================== ================================================
HTTP proxies No (not implemented)
SOCKS proxies No (not supported by the library)

View File

@ -1593,6 +1593,20 @@ Default: ``{}``
A dict containing the pipelines enabled by default in Scrapy. You should never
modify this setting in your project, modify :setting:`ITEM_PIPELINES` instead.
.. setting:: ITEM_PROCESSOR
ITEM_PROCESSOR
--------------
Default: ``"scrapy.pipelines.ItemPipelineManager"``
The :ref:`component <topics-components>` that builds the :ref:`item pipeline
<topics-item-pipeline>` from :setting:`ITEM_PIPELINES` and runs scraped items
through it. It must implement :class:`~scrapy.pipelines.ItemProcessorProtocol`.
.. autoclass:: scrapy.pipelines.ItemProcessorProtocol
:members:
.. setting:: JOBDIR

View File

@ -14,8 +14,8 @@ from twisted.web.client import (
)
from twisted.web.error import SchemeNotSupported
from scrapy.core._http2.protocol import H2ClientFactory, H2ClientProtocol
from scrapy.core.downloader.contextfactory import _AcceptableProtocolsContextFactory
from scrapy.core.http2.protocol import H2ClientFactory, H2ClientProtocol
if TYPE_CHECKING:
from twisted.internet.base import ReactorBase

View File

@ -31,7 +31,7 @@ from twisted.internet.ssl import Certificate
from twisted.protocols.policies import TimeoutMixin
from zope.interface import implementer
from scrapy.core.http2.stream import Stream, StreamCloseReason
from scrapy.core._http2.stream import Stream, StreamCloseReason
from scrapy.exceptions import DownloadTimeoutError, ResponseHeadersTooLargeError
from scrapy.http import Request, Response
from scrapy.utils._download_handlers import get_headers_maxsize_msg

View File

@ -28,7 +28,7 @@ from scrapy.utils.httpobj import urlparse_cached
if TYPE_CHECKING:
from collections.abc import Sequence
from scrapy.core.http2.protocol import H2ClientProtocol
from scrapy.core._http2.protocol import H2ClientProtocol
from scrapy.crawler import Crawler
from scrapy.http import Request, Response

View File

@ -4,9 +4,9 @@ from time import monotonic
from typing import TYPE_CHECKING
from urllib.parse import urldefrag
from scrapy.core._http2.agent import H2Agent, H2ConnectionPool
from scrapy.core.downloader.contextfactory import _load_context_factory_from_settings
from scrapy.core.downloader.handlers.base import BaseDownloadHandler
from scrapy.core.http2.agent import H2Agent, H2ConnectionPool
from scrapy.exceptions import (
DownloadTimeoutError,
NotConfigured,

View File

@ -1,11 +1,12 @@
from __future__ import annotations
import logging
import warnings
from datetime import datetime, timezone
from typing import TYPE_CHECKING, Any
from scrapy import Spider, signals
from scrapy.exceptions import NotConfigured
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.utils.asyncio import AsyncioLoopingCall, create_looping_call
from scrapy.utils.serialize import ScrapyJSONEncoder
@ -38,6 +39,7 @@ class PeriodicLog:
):
self.stats: StatsCollector = stats
self.interval: float = interval
self._multiplier: float = 60.0 / interval
self.task: AsyncioLoopingCall | LoopingCall | None = None
self.encoder: JSONEncoder = ScrapyJSONEncoder(sort_keys=True, indent=4)
self.ext_stats_enabled: bool = bool(ext_stats)
@ -56,6 +58,24 @@ class PeriodicLog:
)
self.ext_timing_enabled: bool = ext_timing_enabled
@property
def multiplier(self) -> float:
warnings.warn(
"The PeriodicLog.multiplier attribute is deprecated.",
ScrapyDeprecationWarning,
stacklevel=2,
)
return self._multiplier
@multiplier.setter
def multiplier(self, value: float) -> None:
warnings.warn(
"The PeriodicLog.multiplier attribute is deprecated.",
ScrapyDeprecationWarning,
stacklevel=2,
)
self._multiplier = value
@classmethod
def from_crawler(cls, crawler: Crawler) -> Self:
interval: float = crawler.settings.getfloat("LOGSTATS_INTERVAL")

View File

@ -8,7 +8,7 @@ from __future__ import annotations
import asyncio
import warnings
from typing import TYPE_CHECKING, Any, cast
from typing import TYPE_CHECKING, Any, Protocol, cast
from twisted.internet.defer import Deferred, DeferredList, FirstError
@ -28,6 +28,23 @@ if TYPE_CHECKING:
from scrapy.settings import Settings
class ItemProcessorProtocol(Protocol):
"""Protocol for item processor implementations.
See :setting:`ITEM_PROCESSOR`.
"""
async def open_spider_async(self) -> None:
"""Get the item processor ready to process items."""
async def process_item_async(self, item: Any) -> Any:
"""Return the processed *item*, or raise
:exc:`~scrapy.exceptions.DropItem` to drop it."""
async def close_spider_async(self) -> None:
"""Release any resource that the item processor is using."""
class ItemPipelineManager(MiddlewareManager):
component_name = "item pipeline"

View File

@ -143,7 +143,7 @@ class TestHttp2(H2DownloadHandlerMixin, TestHttpsBase):
response = await download_handler.download_request(request)
assert response.text == actual_content_length
assert (
"scrapy.core.http2.stream",
"scrapy.core._http2.stream",
logging.WARNING,
f"Ignoring bad Content-Length header "
f"{bad_content_length!r} of request {request}, sending "

View File

@ -7,7 +7,7 @@ from typing import TYPE_CHECKING, Any
import pytest
from scrapy.exceptions import NotConfigured
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
from scrapy.extensions.periodic_log import PeriodicLog
from scrapy.utils.misc import build_from_crawler
from scrapy.utils.test import get_crawler
@ -249,3 +249,17 @@ class TestPeriodicLog:
assert data["time"]["log_interval_real"] >= 0
assert data["time"]["elapsed"] >= 0
assert data["time"]["start_time"] <= data["time"]["utcnow"]
def test_multiplier_deprecated(self) -> None:
crawler = get_crawler(
MetaSpider,
{"PERIODIC_LOG_TIMING_ENABLED": True, "LOGSTATS_INTERVAL": 30},
)
crawler._apply_settings()
ext = build_from_crawler(PeriodicLog, crawler)
with pytest.warns(ScrapyDeprecationWarning):
assert ext.multiplier == 2.0
with pytest.warns(ScrapyDeprecationWarning):
ext.multiplier = 3.0
with pytest.warns(ScrapyDeprecationWarning):
assert ext.multiplier == 3.0

View File

@ -36,7 +36,7 @@ from tests.mockserver.utils import ssl_context_factory
if TYPE_CHECKING:
from collections.abc import AsyncGenerator, Callable, Coroutine, Generator
from scrapy.core.http2.protocol import H2ClientProtocol
from scrapy.core._http2.protocol import H2ClientProtocol
pytestmark = [
@ -242,7 +242,7 @@ class TestHttps2ClientProtocol:
) -> AsyncGenerator[H2ClientProtocol]:
from twisted.internet import reactor
from scrapy.core.http2.protocol import H2ClientFactory # noqa: PLC0415
from scrapy.core._http2.protocol import H2ClientFactory # noqa: PLC0415
client_options = optionsForClientTLS(
hostname=self.host,
@ -460,7 +460,7 @@ class TestHttps2ClientProtocol:
def test_invalid_negotiated_protocol(
self, server_port: int, client: H2ClientProtocol
) -> Generator[Deferred[Any], Any, None]:
with mock.patch("scrapy.core.http2.protocol.PROTOCOL_NAME", new=b"not-h2"):
with mock.patch("scrapy.core._http2.protocol.PROTOCOL_NAME", new=b"not-h2"):
request = Request(url=self.get_url(server_port, "/status?n=200"))
with pytest.raises(ResponseFailed):
yield make_request_dfd(client, request)
@ -527,7 +527,7 @@ class TestHttps2ClientProtocol:
expected_body: bytes,
caplog: pytest.LogCaptureFixture,
) -> None:
with caplog.at_level("WARNING", "scrapy.core.http2.stream"):
with caplog.at_level("WARNING", "scrapy.core._http2.stream"):
response = await make_request(client, request)
assert response.status == 200
assert response.body == expected_body
@ -605,7 +605,7 @@ class TestHttps2ClientProtocol:
def assert_inactive_stream(failure):
assert failure.check(ResponseFailed) is not None
from scrapy.core.http2.stream import InactiveStreamClosed # noqa: PLC0415
from scrapy.core._http2.stream import InactiveStreamClosed # noqa: PLC0415
assert any(
isinstance(e, InactiveStreamClosed) for e in failure.value.reasons
@ -692,7 +692,7 @@ class TestHttps2ClientProtocol:
@staticmethod
async def _check_invalid_netloc(client: H2ClientProtocol, url: str) -> None:
from scrapy.core.http2.stream import InvalidHostname # noqa: PLC0415
from scrapy.core._http2.stream import InvalidHostname # noqa: PLC0415
request = Request(url)
with pytest.raises(InvalidHostname) as exc_info:
@ -737,7 +737,7 @@ class TestHttps2ClientProtocol:
yield make_request_dfd(client, request)
for err in exc_info.value.reasons:
from scrapy.core.http2.protocol import H2ClientProtocol # noqa: PLC0415
from scrapy.core._http2.protocol import H2ClientProtocol # noqa: PLC0415
if isinstance(err, DownloadTimeoutError):
assert (

View File

@ -1641,6 +1641,12 @@ class TestMitmProxyBase(ABC):
assert "Proxy Authentication Required" in log or "407" in log
# Tests below are rerun on failure (see pytest_collection_modifyitems() in the
# root conftest.py), so an attempt must give up soon enough for a rerun to be
# cheap.
REAL_WEBSITE_SETTINGS = {"DOWNLOAD_TIMEOUT": 30}
class TestRealWebsiteBase(ABC):
@property
@abstractmethod
@ -1665,7 +1671,9 @@ class TestRealWebsiteBase(ABC):
async def get_dh(
self, settings_dict: dict[str, Any] | None = None
) -> AsyncGenerator[DownloadHandlerProtocol]:
crawler = get_crawler(DefaultSpider, settings_dict)
crawler = get_crawler(
DefaultSpider, {**REAL_WEBSITE_SETTINGS, **(settings_dict or {})}
)
crawler.spider = crawler._create_spider()
dh = build_from_crawler(self.download_handler_cls, crawler)
try:
@ -1683,7 +1691,9 @@ class TestRealWebsiteBase(ABC):
@coroutine_test
async def test_download_with_spider(self) -> None:
crawler = get_crawler(SingleRequestSpider, self.settings_dict)
crawler = get_crawler(
SingleRequestSpider, {**REAL_WEBSITE_SETTINGS, **(self.settings_dict or {})}
)
await maybe_deferred_to_future(
crawler.crawl(seed=Request("https://books.toscrape.com/"))
)

View File

@ -44,6 +44,7 @@ deps =
pygments
pytest
pytest-cov >= 7.0.0
pytest-rerunfailures
pytest-timeout
pytest-xdist
sybil >= 1.3.0 # https://github.com/cjw296/sybil/issues/20#issuecomment-605433422
@ -124,7 +125,7 @@ commands =
[testenv:twinecheck]
basepython = python3
deps =
twine==6.2.0
twine==7.0.0
build==1.5.0
commands =
python -m build --sdist