From a0672532342ae63110557c01cb20b82bc8adc60e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 13 Mar 2025 08:58:12 +0100 Subject: [PATCH] Allow overriding the active seeding policy --- docs/news.rst | 26 +++++++++++- docs/topics/settings.rst | 9 +++-- docs/topics/spider-middleware.rst | 21 ++++++++-- scrapy/core/_seeding.py | 2 +- scrapy/core/engine.py | 67 +++++++++++++++++++++---------- scrapy/spiders/__init__.py | 13 +++++- tests/test_engine_seeding.py | 66 +++++++++++++++++++++++++++++- 7 files changed, 169 insertions(+), 35 deletions(-) diff --git a/docs/news.rst b/docs/news.rst index 7afcbb3cf..d7de9c769 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -40,14 +40,16 @@ Deprecations use :meth:`~scrapy.Spider.yield_seeds` instead, or both to maintain support for lower Scrapy versions. - (:issue:`456`, :issue:`3477`, :issue:`4467`, :issue:`5627`) + (:issue:`456`, :issue:`3477`, :issue:`4467`, :issue:`5627`, :issue:`6715`, + :issue:`6729`) - The ``process_start_requests()`` method of :ref:`spider middlewares ` is deprecated, use :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_seeds` instead, or both to maintain support for lower Scrapy versions. - (:issue:`456`, :issue:`3477`, :issue:`4467`, :issue:`5627`) + (:issue:`456`, :issue:`3477`, :issue:`4467`, :issue:`5627`, :issue:`6715`, + :issue:`6729`) New features ~~~~~~~~~~~~ @@ -66,6 +68,26 @@ New features - The new :setting:`SEEDING_POLICY` setting allows customizing how spider start requests and items are consumed. + You can also override the active seeding policy from + :meth:`Spider.yield_seeds ` and from + :meth:`SpiderMiddleware.process_seeds + `. + + .. note:: Some third-party spider middlewares may need to be updated for + Scrapy VERSION support before you can use them in combination with the + ability to override the active seeding policy. + + (:issue:`740`, :issue:`1051`, :issue:`1443`, :issue:`3237`, :issue:`4467`, + :issue:`5282`, :issue:`6715`) + +Bug fixes +~~~~~~~~~ + +- The first :setting:`CONCURRENT_REQUESTS` start requests are no longer sent + in reserve order by default. + + (:issue:`6715`, :issue:`6729`) + .. _release-2.12.0: diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index dd2fb3400..d773757d0 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1746,9 +1746,12 @@ Determines the way :meth:`Spider.yield_seeds ` is iterated. Its value may be defined as a member of the :class:`~scrapy.SeedingPolicy` enum -(e.g. :py:enum:mem:`SeedingPolicy.front_load -`) or as the corresponding string (e.g. -``"front-load"``). +(e.g. :py:enum:mem:`SeedingPolicy.lazy `) or as a +matching string (e.g. ``"lazy"``). + +You can also override the active seeding policy from :meth:`Spider.yield_seeds +` and from :meth:`SpiderMiddleware.process_seeds +`. .. autoenum:: scrapy.SeedingPolicy :members: diff --git a/docs/topics/spider-middleware.rst b/docs/topics/spider-middleware.rst index 36b440720..9ea46fc53 100644 --- a/docs/topics/spider-middleware.rst +++ b/docs/topics/spider-middleware.rst @@ -83,12 +83,25 @@ one or more of these methods: async for seed in seeds: yield seed - You may yield :class:`~scrapy.Request` or :ref:`item ` - objects, same as :meth:`~scrapy.Spider.yield_seeds`, from *seeds* or - not. + You may yield the same type of objects as + :meth:`~scrapy.Spider.yield_seeds`. As with :meth:`~scrapy.Spider.yield_seeds`, how this method is iterated - is controlled by :setting:`SEEDING_POLICY`. + by default is controlled by :setting:`SEEDING_POLICY`. It is also + possible to yield a :class:`~scrapy.SeedingPolicy` enum or a matching + string to change the active seeding policy, for example: + + .. code-block:: python + + async def process_seeds(self, seeds): + yield "front_load" + async for seed in seeds: + yield seed + yield "idle" + + .. tip:: You can also restore the configured seeding policy by + :ref:`reading its value ` from the + :setting:`SEEDING_POLICY` setting and yielding it. To write spider middlewares that work on Scrapy versions lower than VERSION, define also a synchronous ``process_start_requests()`` method diff --git a/scrapy/core/_seeding.py b/scrapy/core/_seeding.py index 13f5c6d6a..83f3f4159 100644 --- a/scrapy/core/_seeding.py +++ b/scrapy/core/_seeding.py @@ -15,7 +15,7 @@ else: @document_enum class SeedingPolicy(Enum): - front_load = "front-load" + front_load = "front_load" """The crawl does not start until all seed requests have been scheduled. Aims to give the :ref:`scheduler ` full control over diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 171b89434..8383b9395 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -43,6 +43,10 @@ logger = logging.getLogger(__name__) _T = TypeVar("_T") +class _SeedingPolicyChange(Exception): + pass + + class _Slot: def __init__( self, @@ -204,6 +208,21 @@ class ExecutionEngine: and not self._needs_backout() ): self._start_scheduled_request() + elif isinstance(seed, (str, SeedingPolicy)): + try: + self._seeding_policy = SeedingPolicy(seed) + except ValueError: + valid_policy_strings = ", ".join( + policy.value for policy in SeedingPolicy + ) + logger.error( + f"Seed {seed!r} has been ignored. Seeds of {str} type " + f"must be valid seeding policies " + f"({valid_policy_strings})." + ) + self._slot.nextcall.schedule() + else: + raise _SeedingPolicyChange else: self.scraper.start_itemproc(seed, response=None) self._slot.nextcall.schedule() @@ -217,31 +236,35 @@ class ExecutionEngine: if self._slot is None or self._slot.closing is not None or self.paused: return - if self._seeding_policy in {SeedingPolicy.idle, SeedingPolicy.lazy}: - while not self._needs_backout(): - if self._start_scheduled_request() is None: - break - if ( - self._seeds is not None - and not self._needs_backout() - and ( - self._seeding_policy is not SeedingPolicy.idle - or (not self._waiting_for_seed and not self.downloader.active) - ) - ): - yield self._process_next_seed() - else: - assert self._seeding_policy in { - SeedingPolicy.front_load, - SeedingPolicy.greedy, - } - if self._seeds is not None: - if not self._needs_backout(): - yield self._process_next_seed() - else: + try: + if self._seeding_policy in {SeedingPolicy.idle, SeedingPolicy.lazy}: while not self._needs_backout(): if self._start_scheduled_request() is None: break + if ( + self._seeds is not None + and not self._needs_backout() + and ( + self._seeding_policy is not SeedingPolicy.idle + or (not self._waiting_for_seed and not self.downloader.active) + ) + ): + yield self._process_next_seed() + else: + assert self._seeding_policy in { + SeedingPolicy.front_load, + SeedingPolicy.greedy, + } + if self._seeds is not None: + if not self._needs_backout(): + yield self._process_next_seed() + else: + while not self._needs_backout(): + if self._start_scheduled_request() is None: + break + except _SeedingPolicyChange: + self._slot.nextcall.schedule() + return if self.spider_is_idle() and self._slot.close_if_idle: self._spider_idle() diff --git a/scrapy/spiders/__init__.py b/scrapy/spiders/__init__.py index 9ff491944..d2941c7c3 100644 --- a/scrapy/spiders/__init__.py +++ b/scrapy/spiders/__init__.py @@ -114,7 +114,18 @@ class Spider(object_ref): yield {"foo": "bar"} Use :setting:`SEEDING_POLICY` to set how :meth:`yield_seeds` is - iterated. + iterated by default. It is also + possible to yield a :class:`~scrapy.SeedingPolicy` enum or a matching + string to change the active seeding policy, for example: + + .. code-block:: python + + async def yield_seeds(self): + yield "front_load" + yield Request("https://a.example") + yield Request("https://b.example") + yield self.crawler.settings["SEEDING_POLICY"] + yield Request("https://c.example") To write spiders that work on Scrapy versions lower than VERSION, define also a synchronous ``start_requests()`` method that returns an diff --git a/tests/test_engine_seeding.py b/tests/test_engine_seeding.py index 4debd90a0..9fbe4c6e7 100644 --- a/tests/test_engine_seeding.py +++ b/tests/test_engine_seeding.py @@ -1,10 +1,12 @@ from __future__ import annotations from collections import defaultdict, deque +from logging import ERROR +from testfixtures import LogCapture from twisted.trial.unittest import TestCase -from scrapy import Request, Spider, signals +from scrapy import Request, SeedingPolicy, Spider, signals from scrapy.core.engine import ExecutionEngine from scrapy.core.scheduler import BaseScheduler from scrapy.utils.defer import deferred_f_from_coro_f, maybe_deferred_to_future @@ -221,7 +223,7 @@ class MainTestCase(TestCase): def track_url(request, spider): actual_urls.append(request.url) - settings = {"SCHEDULER": TestScheduler, "SEEDING_POLICY": "front-load"} + settings = {"SCHEDULER": TestScheduler, "SEEDING_POLICY": "front_load"} crawler = get_crawler(TestSpider, settings_dict=settings) crawler.signals.connect(track_url, signals.request_reached_downloader) await maybe_deferred_to_future(crawler.crawl()) @@ -229,6 +231,66 @@ class MainTestCase(TestCase): expected_urls = ["data:,a", "data:,b"] assert actual_urls == expected_urls, f"{actual_urls=} != {expected_urls=}" + @deferred_f_from_coro_f + async def test_override(self): + class TestScheduler(BaseScheduler): + def __init__(self, *args, **kwargs): + self.requests = defaultdict(deque) + + def enqueue_request(self, request: Request) -> bool: + self.requests[request.priority].append(request) + return True + + def has_pending_requests(self) -> bool: + return bool(self.requests) + + def next_request(self) -> Request | None: + if not self.requests: + return None + priority = max(self.requests) + request = self.requests[priority].popleft() + if not self.requests[priority]: + del self.requests[priority] + return request + + class TestSpider(Spider): + name = "test" + + async def yield_seeds(self): + yield "front-load" # typo + yield SeedingPolicy.front_load + yield Request("data:,b", priority=1) + yield Request("data:,a", priority=2) + yield self.crawler.settings["SEEDING_POLICY"] + yield Request("data:,c", priority=3) + + def parse(self, response): + pass + + actual_items = [] + actual_urls = [] + + def track_item(item, response, spider): + actual_items.append(item) + + def track_url(request, spider): + actual_urls.append(request.url) + + settings = {"SCHEDULER": TestScheduler} + crawler = get_crawler(TestSpider, settings_dict=settings) + crawler.signals.connect(track_item, signals.item_scraped) + crawler.signals.connect(track_url, signals.request_reached_downloader) + with LogCapture(level=ERROR) as log: + await maybe_deferred_to_future(crawler.crawl()) + assert len(log.records) == 1 + assert "must be valid seeding policies" in str(log.records[0]) + assert crawler.stats.get_value("finish_reason") == "finished" + assert not actual_items, ( + f"{actual_items=} should be empty, policies are not items" + ) + expected_urls = ["data:,a", "data:,b", "data:,c"] + assert actual_urls == expected_urls, f"{actual_urls=} != {expected_urls=}" + class MockServerTestCase(TestCase): # See the comment on the matching line above.