From 67e52826845f29dedf33a21ae9d83211e4867be6 Mon Sep 17 00:00:00 2001 From: Adrian Date: Thu, 23 Jul 2026 13:02:41 +0200 Subject: [PATCH] Use autodoc for exceptions and improve their docs (#7767) --- docs/topics/exceptions.rst | 110 +++---------------------------------- scrapy/exceptions.py | 75 +++++++++++++++++++++---- 2 files changed, 74 insertions(+), 111 deletions(-) diff --git a/docs/topics/exceptions.rst b/docs/topics/exceptions.rst index 45acb1763..0aab90a43 100644 --- a/docs/topics/exceptions.rst +++ b/docs/topics/exceptions.rst @@ -1,117 +1,25 @@ .. _topics-exceptions: +.. _topics-exceptions-ref: ========== Exceptions ========== -.. module:: scrapy.exceptions - :synopsis: Scrapy exceptions - -.. _topics-exceptions-ref: - -Built-in Exceptions reference -============================= - Here's a list of all exceptions included in Scrapy and their usage, except for the :ref:`download handler exceptions `. +.. module:: scrapy.exceptions -CloseSpider ------------ +.. autoexception:: CloseSpider -.. exception:: CloseSpider(reason='cancelled') +.. autoexception:: DontCloseSpider - This exception can be raised from a spider callback to request the spider to be - closed/stopped. Supported arguments: +.. autoexception:: DropItem - :param reason: the reason for closing - :type reason: str +.. autoexception:: IgnoreRequest -For example: +.. autoexception:: NotConfigured -.. code-block:: python +.. autoexception:: NotSupported - def parse_page(self, response): - if "Bandwidth exceeded" in response.text: - raise CloseSpider("bandwidth_exceeded") - -DontCloseSpider ---------------- - -.. exception:: DontCloseSpider - -This exception can be raised in a :signal:`spider_idle` signal handler to -prevent the spider from being closed. - -DropItem --------- - -.. exception:: DropItem - -The exception that must be raised by item pipeline stages to stop processing an -Item. For more information see :ref:`topics-item-pipeline`. - -IgnoreRequest -------------- - -.. exception:: IgnoreRequest - -This exception can be raised by the Scheduler or any downloader middleware to -indicate that the request should be ignored. - -NotConfigured -------------- - -.. exception:: NotConfigured - -This exception can be raised by some components to indicate that they will -remain disabled. Those components include: - -- Extensions -- Item pipelines -- Downloader middlewares -- Spider middlewares - -The exception must be raised in the component's ``__init__()`` or -``from_crawler()`` method. - -NotSupported ------------- - -.. exception:: NotSupported - -This exception is raised to indicate an unsupported feature. - -StopDownload -------------- - -.. exception:: StopDownload(fail=True) - -Raised from a :class:`~scrapy.signals.bytes_received` or :class:`~scrapy.signals.headers_received` -signal handler to indicate that no further bytes should be downloaded for a response. - -The ``fail`` boolean parameter controls which method will handle the resulting -response: - -* If ``fail=True`` (default), the request errback is called. The response object is - available as the ``response`` attribute of the ``StopDownload`` exception, - which is in turn stored as the ``value`` attribute of the received - :class:`~twisted.python.failure.Failure` object. This means that in an errback - defined as ``def errback(self, failure)``, the response can be accessed though - ``failure.value.response``. - -* If ``fail=False``, the request callback is called instead. - -In both cases, the response could have its body truncated: the body contains -all bytes received up until the exception is raised, including the bytes -received in the signal handler that raises the exception. Also, the response -object is marked with ``"download_stopped"`` in its :attr:`~scrapy.http.Response.flags` -attribute. - -.. note:: ``fail`` is a keyword-only parameter, i.e. raising - ``StopDownload(False)`` or ``StopDownload(True)`` will raise - a :class:`TypeError`. - -See the documentation for the :class:`~scrapy.signals.bytes_received` and -:class:`~scrapy.signals.headers_received` signals -and the :ref:`topics-stop-response-download` topic for additional information and examples. +.. autoexception:: StopDownload diff --git a/scrapy/exceptions.py b/scrapy/exceptions.py index 5330eab48..ccaccddf2 100644 --- a/scrapy/exceptions.py +++ b/scrapy/exceptions.py @@ -16,7 +16,15 @@ if TYPE_CHECKING: class NotConfigured(Exception): - """Indicates a missing configuration situation""" + """Raised by a :ref:`component ` from its ``__init__()`` + or :meth:`from_crawler` method to indicate that it will remain disabled. + + Only the following components can be disabled this way: + + - :ref:`Downloader middlewares ` + - :ref:`Extensions ` + - :ref:`Item pipelines ` + - :ref:`Spider middlewares `""" class _InvalidOutput(TypeError): @@ -30,15 +38,37 @@ class _InvalidOutput(TypeError): class IgnoreRequest(Exception): - """Indicates a decision was made not to process a request""" + """Raised to indicate that a request should be ignored. + + A :ref:`downloader middleware ` can raise it + from its + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_request` + or + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_response` + method to drop a request, and a :signal:`request_scheduled` signal handler + can raise it to drop a request before it reaches the + :ref:`scheduler `.""" class DontCloseSpider(Exception): - """Request the spider not to be closed yet""" + """Raised in a :signal:`spider_idle` signal handler to prevent the spider + from being closed.""" class CloseSpider(Exception): - """Raise this from callbacks to request the spider to be closed""" + """Raised from a :ref:`spider callback ` to request the + spider to be closed/stopped. + + *reason* is a string with the reason for closing. + + For example: + + .. code-block:: python + + def parse_page(self, response): + if "Bandwidth exceeded" in response.text: + raise CloseSpider("bandwidth_exceeded") + """ def __init__(self, reason: str = "cancelled"): super().__init__() @@ -46,10 +76,27 @@ class CloseSpider(Exception): class StopDownload(Exception): - """ - Stop the download of the body for a given response. - The 'fail' boolean parameter indicates whether or not the resulting partial response - should be handled by the request errback. Note that 'fail' is a keyword-only argument. + """Raised from a :class:`~scrapy.signals.bytes_received` or + :class:`~scrapy.signals.headers_received` signal handler to :ref:`stop the + download ` of the response body. + + The ``fail`` boolean parameter controls which method will handle the + resulting response: + + * If ``fail=True`` (default), the request errback is called. The response + object is available as the ``response`` attribute of the ``StopDownload`` + exception, which is in turn stored as the ``value`` attribute of the + received :class:`~twisted.python.failure.Failure` object. This means that + in an errback defined as ``def errback(self, failure)``, the response can + be accessed though ``failure.value.response``. + + * If ``fail=False``, the request callback is called instead. + + In both cases, the response could have its body truncated: the body contains + all bytes received up until the exception is raised, including the bytes + received in the signal handler that raises the exception. Also, the response + object is marked with ``"download_stopped"`` in its + :attr:`~scrapy.http.Response.flags` attribute. """ response: Response | None @@ -91,7 +138,8 @@ class UnsupportedURLSchemeError(Exception): class DropItem(Exception): - """Drop item from the item pipeline""" + """Raised from the :meth:`process_item` method of an :ref:`item pipeline + ` to stop the processing of an item.""" def __init__(self, message: str, log_level: str | None = None): super().__init__(message) @@ -99,7 +147,14 @@ class DropItem(Exception): class NotSupported(Exception): - """Indicates a feature or method is not supported""" + """Raised to indicate that a requested feature is not supported. + + For example, Scrapy raises it when text-parsing shortcuts such as + :meth:`response.css() ` or + :meth:`response.xpath() ` are used on a + :class:`~scrapy.http.Response` whose content is not text, or when sending a + request whose URL scheme has no matching :ref:`download handler + `.""" # Commands