tests(memusage): integration coverage with CI-safe reactor pinning and log assertions

This commit is contained in:
Ayman 2025-08-23 02:04:12 +01:00
parent fdab0220d2
commit edf872fcb2
1 changed files with 27 additions and 17 deletions

View File

@ -16,7 +16,7 @@ from scrapy.extensions.memusage import MemoryUsage
from scrapy.settings import Settings
from scrapy.spiders import Spider
# Skip on Windows; memusage relies on 'resource'
# Memusage relies on 'resource' (Unix only).
pytestmark = pytest.mark.skipif(
sys.platform.startswith("win"),
reason="MemoryUsage extension not available on Windows",
@ -24,7 +24,7 @@ pytestmark = pytest.mark.skipif(
class _LoopSpider(Spider):
"""Keep the crawl alive long enough for periodic checks to run."""
"""Keeps the engine running long enough for periodic checks."""
name = "loop-file-spider"
@ -44,10 +44,23 @@ class _LoopSpider(Spider):
def _tmp_file_uri(tmp_path: Path) -> str:
f = tmp_path / "hello.txt"
f.write_text("hello\\n")
f.write_text("hello\n")
return f.as_uri()
def _pin_reactor_to_installed(settings: Settings) -> None:
"""Honor the reactor already installed by the test env/CI."""
from twisted.internet import (
reactor as _reactor, # local import to avoid import-time issues
)
settings.set(
"TWISTED_REACTOR",
f"{_reactor.__class__.__module__}.{_reactor.__class__.__name__}",
priority="cmdline",
)
@inlineCallbacks
@pytest.mark.twisted
def test_memusage_limit_closes_spider_with_reason_and_error_log(
@ -62,8 +75,8 @@ def test_memusage_limit_closes_spider_with_reason_and_error_log(
"LOG_LEVEL": "INFO",
}
)
_pin_reactor_to_installed(settings)
# Start LOW, flip HIGH only after spider is opened.
MB = 1024 * 1024
state = {"high": False}
@ -75,19 +88,18 @@ def test_memusage_limit_closes_spider_with_reason_and_error_log(
runner = CrawlerRunner(settings)
crawler = runner.create_crawler(_LoopSpider)
# Use the correct kwarg name from the signal: 'spider'
def on_opened(spider):
state["high"] = True
crawler.signals.connect(on_opened, signal=signals.spider_opened)
caplog.set_level(logging.ERROR, logger="scrapy.extensions.memusage")
yield runner.crawl(crawler, url=url, loops=60) # plenty of time for checks
yield runner.crawl(crawler, url=url, loops=100)
# Assert finish reason via stats (black-box)
assert crawler.stats.get_value("finish_reason") == "memusage_exceeded"
# Assert the ERROR log message was emitted
assert any("memory usage exceeded" in r.message.lower() for r in caplog.records)
assert any(
"memory usage exceeded" in r.getMessage().lower() for r in caplog.records
)
@inlineCallbacks
@ -98,28 +110,26 @@ def test_memusage_warning_logs_but_allows_normal_finish(tmp_path, caplog, monkey
{
"MEMUSAGE_ENABLED": True,
"MEMUSAGE_WARNING_MB": 50,
"MEMUSAGE_LIMIT_MB": 0, # no hard limit
"MEMUSAGE_LIMIT_MB": 0,
"MEMUSAGE_CHECK_INTERVAL_SECONDS": 0.01,
"LOG_LEVEL": "INFO",
}
)
_pin_reactor_to_installed(settings)
MB = 1024 * 1024
# Always above warning, never limited
monkeypatch.setattr(MemoryUsage, "get_virtual_size", lambda self: 75 * MB)
runner = CrawlerRunner(settings)
crawler = runner.create_crawler(_LoopSpider)
caplog.set_level(logging.WARNING, logger="scrapy.extensions.memusage")
yield runner.crawl(crawler, url=url, loops=40)
yield runner.crawl(crawler, url=url, loops=60)
# Normal completion
assert crawler.stats.get_value("finish_reason") == "finished"
# Warning log appeared (match actual message)
assert any(
"memory usage reached" in r.message.lower()
or "memory usage warning" in r.message.lower()
or "warning: memory usage reached" in r.message.lower()
("memory usage reached" in r.getMessage().lower())
or ("memory usage warning" in r.getMessage().lower())
or ("warning: memory usage reached" in r.getMessage().lower())
for r in caplog.records
)