mirror of https://github.com/scrapy/scrapy.git
Extend benchmarks (#7887)
This commit is contained in:
parent
91b70e4db4
commit
e0128c20c3
|
|
@ -1,14 +1,53 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from scrapy.http import Response
|
||||
from scrapy.utils.test import get_crawler
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from scrapy import Spider
|
||||
from scrapy import Request, Spider
|
||||
from scrapy.crawler import Crawler
|
||||
|
||||
|
||||
class NullDownloadHandler:
|
||||
"""Download handler that returns an empty response without doing any I/O.
|
||||
|
||||
It lets benchmarks measure the engine, the scheduler and the middlewares
|
||||
without also measuring HTTP parsing and socket handling, and reach as many
|
||||
hostnames as they need without DNS resolution.
|
||||
|
||||
It yields control to the event loop once per request, so that requests can
|
||||
be in progress at the same time and concurrency limits apply. The peak
|
||||
number of requests in progress is tracked in the
|
||||
``benchmark/peak_concurrency`` stat.
|
||||
"""
|
||||
|
||||
lazy = False
|
||||
|
||||
def __init__(self, crawler: Crawler):
|
||||
self._crawler = crawler
|
||||
self._active = 0
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler: Crawler) -> NullDownloadHandler:
|
||||
return cls(crawler)
|
||||
|
||||
async def download_request(self, request: Request) -> Response:
|
||||
self._active += 1
|
||||
assert self._crawler.stats
|
||||
self._crawler.stats.max_value("benchmark/peak_concurrency", self._active)
|
||||
try:
|
||||
await asyncio.sleep(0)
|
||||
return Response(request.url, request=request)
|
||||
finally:
|
||||
self._active -= 1
|
||||
|
||||
async def close(self) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def crawl(spidercls: type[Spider], settings: dict[str, Any], **kwargs: Any) -> Crawler:
|
||||
"""Run a crawl to completion and return its crawler.
|
||||
|
||||
|
|
|
|||
|
|
@ -7,13 +7,14 @@ import pytest
|
|||
|
||||
from scrapy import Field, Item, Request, Spider
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
from tests.benchmarks import crawl
|
||||
from tests.benchmarks import NullDownloadHandler, crawl
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from collections.abc import AsyncIterator
|
||||
|
||||
from pytest_codspeed import BenchmarkFixture # type: ignore[import-not-found]
|
||||
|
||||
from scrapy.crawler import Crawler
|
||||
from scrapy.http import Response
|
||||
from tests.mockserver.http import MockServer
|
||||
|
||||
|
|
@ -22,6 +23,22 @@ pytest.importorskip("pytest_codspeed", reason="Benchmarks require pytest-codspee
|
|||
PAGES = 100
|
||||
LINKS_PER_PAGE = 5
|
||||
|
||||
# Requests per crawl of the benchmarks that use NullDownloadHandler. The broad
|
||||
# crawl scenarios split them differently between hostnames and pages per
|
||||
# hostname.
|
||||
REQUESTS = 200
|
||||
BROAD_DEEP_PAGES = 10
|
||||
|
||||
# Requests per crawl and delay of the benchmark that measures delayed requests,
|
||||
# where wall time, unlike in the other benchmarks, is a function of the delay.
|
||||
DELAYED_REQUESTS = 50
|
||||
DELAY = 0.005
|
||||
|
||||
NULL_SETTINGS: dict[str, Any] = {
|
||||
"DOWNLOAD_HANDLERS": {"http": NullDownloadHandler},
|
||||
"LOG_ENABLED": False,
|
||||
}
|
||||
|
||||
|
||||
class _Page(Item):
|
||||
url = Field()
|
||||
|
|
@ -45,11 +62,43 @@ class _FollowSpider(Spider):
|
|||
yield Request(link.url)
|
||||
|
||||
|
||||
class _TreeSpider(Spider):
|
||||
"""Crawl *pages* pages on each of *domains* hostnames.
|
||||
|
||||
Pages are numbered from 1, and page *n* links to pages *2n* and *2n+1*, so
|
||||
that requests also reach the scheduler from callbacks, and not only from
|
||||
:meth:`~scrapy.Spider.start`.
|
||||
"""
|
||||
|
||||
name = "benchmark-tree"
|
||||
domains: int = 1
|
||||
pages: int = 1
|
||||
|
||||
async def start(self) -> AsyncIterator[Any]:
|
||||
for domain in range(self.domains):
|
||||
yield Request(f"http://d{domain}.example.com/1")
|
||||
|
||||
def parse(self, response: Response) -> Any:
|
||||
page = int(response.url.rpartition("/")[2])
|
||||
for child in (page * 2, page * 2 + 1):
|
||||
if child <= self.pages:
|
||||
yield Request(response.urljoin(f"/{child}"))
|
||||
|
||||
|
||||
class _Pipeline:
|
||||
def process_item(self, item: Any) -> Any:
|
||||
return item
|
||||
|
||||
|
||||
def _crawl_tree(settings: dict[str, Any], *, domains: int, pages: int) -> Crawler:
|
||||
crawler = crawl(
|
||||
_TreeSpider, {**NULL_SETTINGS, **settings}, domains=domains, pages=pages
|
||||
)
|
||||
assert crawler.stats
|
||||
assert crawler.stats.get_value("downloader/response_count") == domains * pages
|
||||
return crawler
|
||||
|
||||
|
||||
def test_overhead_http(benchmark: BenchmarkFixture, mockserver: MockServer) -> None:
|
||||
"""Per-request overhead of a crawl over HTTP.
|
||||
|
||||
|
|
@ -67,3 +116,50 @@ def test_overhead_http(benchmark: BenchmarkFixture, mockserver: MockServer) -> N
|
|||
assert crawler.stats.get_value("item_scraped_count") == PAGES + 1
|
||||
|
||||
benchmark(run)
|
||||
|
||||
|
||||
def test_overhead_engine(benchmark: BenchmarkFixture) -> None:
|
||||
"""Per-request overhead of a crawl of a single hostname without any I/O."""
|
||||
|
||||
def run() -> None:
|
||||
crawler = _crawl_tree({}, domains=1, pages=REQUESTS)
|
||||
assert crawler.stats
|
||||
assert crawler.stats.get_value("benchmark/peak_concurrency") > 1
|
||||
|
||||
benchmark(run)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("domains", "pages"),
|
||||
[
|
||||
pytest.param(REQUESTS, 1, id="shallow"),
|
||||
pytest.param(REQUESTS // BROAD_DEEP_PAGES, BROAD_DEEP_PAGES, id="deep"),
|
||||
],
|
||||
)
|
||||
def test_overhead_broad(benchmark: BenchmarkFixture, domains: int, pages: int) -> None:
|
||||
"""Per-request overhead of a broad crawl.
|
||||
|
||||
The shallow scenario, which reaches a single page of every hostname, pays
|
||||
the cost of tracking a hostname for the first time on every request, and
|
||||
gets its requests from :meth:`~scrapy.Spider.start`. The deep scenario,
|
||||
which reaches the same number of pages spread over fewer hostnames,
|
||||
amortizes that cost, and instead keeps several requests per hostname
|
||||
waiting in the scheduler.
|
||||
"""
|
||||
benchmark(lambda: _crawl_tree({}, domains=domains, pages=pages))
|
||||
|
||||
|
||||
def test_overhead_concurrency(benchmark: BenchmarkFixture) -> None:
|
||||
"""Overhead of a crawl limited to 1 request at a time on a single hostname."""
|
||||
settings = {"CONCURRENT_REQUESTS_PER_DOMAIN": 1}
|
||||
benchmark(lambda: _crawl_tree(settings, domains=1, pages=REQUESTS))
|
||||
|
||||
|
||||
def test_overhead_delay(benchmark: BenchmarkFixture) -> None:
|
||||
"""Overhead of a crawl where every request waits for a download delay.
|
||||
|
||||
The delay is not randomized, so that wall time, and hence the number of
|
||||
reactor iterations that the crawl needs, does not change between runs.
|
||||
"""
|
||||
settings = {"DOWNLOAD_DELAY": DELAY, "RANDOMIZE_DOWNLOAD_DELAY": False}
|
||||
benchmark(lambda: _crawl_tree(settings, domains=1, pages=DELAYED_REQUESTS))
|
||||
|
|
|
|||
Loading…
Reference in New Issue