diff --git a/docs/conf.py b/docs/conf.py index e2784cf17..0e0df274c 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -276,5 +276,6 @@ coverage_ignore_pyobjects = [ intersphinx_mapping = { 'python': ('https://docs.python.org/3', None), - 'sphinx': ('https://www.sphinx-doc.org/en/stable', None), + 'sphinx': ('https://www.sphinx-doc.org/en/master', None), + 'twisted': ('https://twistedmatrix.com/documents/current', None), } diff --git a/docs/contributing.rst b/docs/contributing.rst index f084bd23d..68ae2bf3c 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -194,8 +194,9 @@ documentation instead of duplicating the docstring in files within the Tests ===== -Tests are implemented using the `Twisted unit-testing framework`_, running -tests requires `tox`_. +Tests are implemented using the :doc:`Twisted unit-testing framework +`. Running tests requires +`tox`_. .. _running-tests: @@ -269,7 +270,6 @@ And their unit-tests are in:: .. _issue tracker: https://github.com/scrapy/scrapy/issues .. _scrapy-users: https://groups.google.com/forum/#!forum/scrapy-users .. _Scrapy subreddit: https://reddit.com/r/scrapy -.. _Twisted unit-testing framework: https://twistedmatrix.com/documents/current/core/development/policy/test-standard.html .. _AUTHORS: https://github.com/scrapy/scrapy/blob/master/AUTHORS .. _tests/: https://github.com/scrapy/scrapy/tree/master/tests .. _open issues: https://github.com/scrapy/scrapy/issues diff --git a/docs/topics/api.rst b/docs/topics/api.rst index 7c8c40b5f..1c461a511 100644 --- a/docs/topics/api.rst +++ b/docs/topics/api.rst @@ -273,5 +273,3 @@ class (which they all inherit from). Close the given spider. After this is called, no more specific stats can be accessed or collected. - -.. _reactor: https://twistedmatrix.com/documents/current/core/howto/reactor-basics.html diff --git a/docs/topics/architecture.rst b/docs/topics/architecture.rst index 2effe94dc..ae25dfa2f 100644 --- a/docs/topics/architecture.rst +++ b/docs/topics/architecture.rst @@ -166,11 +166,10 @@ for concurrency. For more information about asynchronous programming and Twisted see these links: -* `Introduction to Deferreds in Twisted`_ +* :doc:`twisted:core/howto/defer-intro` * `Twisted - hello, asynchronous programming`_ * `Twisted Introduction - Krondo`_ .. _Twisted: https://twistedmatrix.com/trac/ -.. _Introduction to Deferreds in Twisted: https://twistedmatrix.com/documents/current/core/howto/defer-intro.html .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _Twisted Introduction - Krondo: http://krondo.com/an-introduction-to-asynchronous-programming-and-twisted/ diff --git a/docs/topics/email.rst b/docs/topics/email.rst index 12eedf2cd..72bf52227 100644 --- a/docs/topics/email.rst +++ b/docs/topics/email.rst @@ -9,13 +9,13 @@ Sending e-mail Although Python makes sending e-mails relatively easy via the `smtplib`_ library, Scrapy provides its own facility for sending e-mails which is very -easy to use and it's implemented using `Twisted non-blocking IO`_, to avoid -interfering with the non-blocking IO of the crawler. It also provides a -simple API for sending attachments and it's very easy to configure, with a few -:ref:`settings `. +easy to use and it's implemented using :doc:`Twisted non-blocking IO +`, to avoid interfering with the non-blocking +IO of the crawler. It also provides a simple API for sending attachments and +it's very easy to configure, with a few :ref:`settings +`. .. _smtplib: https://docs.python.org/2/library/smtplib.html -.. _Twisted non-blocking IO: https://twistedmatrix.com/documents/current/core/howto/defer-intro.html Quick example ============= @@ -39,7 +39,8 @@ MailSender class reference ========================== MailSender is the preferred class to use for sending emails from Scrapy, as it -uses `Twisted non-blocking IO`_, like the rest of the framework. +uses :doc:`Twisted non-blocking IO `, like the +rest of the framework. .. class:: MailSender(smtphost=None, mailfrom=None, smtpuser=None, smtppass=None, smtpport=None) diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index fae18200a..cdc4953c2 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -29,7 +29,8 @@ Each item pipeline component is a Python class that must implement the following This method is called for every item pipeline component. :meth:`process_item` must either: return a dict with data, return an :class:`~scrapy.item.Item` - (or any descendant class) object, return a `Twisted Deferred`_ or raise + (or any descendant class) object, return a + :class:`~twisted.internet.defer.Deferred` or raise :exc:`~scrapy.exceptions.DropItem` exception. Dropped items are no longer processed by further pipeline components. @@ -67,8 +68,6 @@ Additionally, they may also implement the following methods: :type crawler: :class:`~scrapy.crawler.Crawler` object -.. _Twisted Deferred: https://twistedmatrix.com/documents/current/core/howto/defer.html - Item pipeline example ===================== @@ -166,7 +165,8 @@ method and how to clean up the resources properly.:: Take screenshot of item ----------------------- -This example demonstrates how to return Deferred_ from :meth:`process_item` method. +This example demonstrates how to return a +:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method. It uses Splash_ to render screenshot of item url. Pipeline makes request to locally running instance of Splash_. After request is downloaded and Deferred callback fires, it saves item to a file and adds filename to an item. @@ -209,7 +209,6 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite return item .. _Splash: https://splash.readthedocs.io/en/stable/ -.. _Deferred: https://twistedmatrix.com/documents/current/core/howto/defer.html Duplicates filter ----------------- diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 431cc6027..206e7cfa5 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -441,8 +441,9 @@ See here the methods that you can override in your custom Files Pipeline: * ``success`` is a boolean which is ``True`` if the image was downloaded successfully or ``False`` if it failed for some reason - * ``file_info_or_error`` is a dict containing the following keys (if success - is ``True``) or a `Twisted Failure`_ if there was a problem. + * ``file_info_or_error`` is a dict containing the following keys (if + success is ``True``) or a :exc:`~twisted.python.failure.Failure` if + there was a problem. * ``url`` - the url where the file was downloaded from. This is the url of the request returned from the :meth:`~get_media_requests` @@ -577,5 +578,4 @@ above:: item['image_paths'] = image_paths return item -.. _Twisted Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html .. _MD5 hash: https://en.wikipedia.org/wiki/MD5 diff --git a/docs/topics/practices.rst b/docs/topics/practices.rst index a6d4f0d6d..e3e8fdc72 100644 --- a/docs/topics/practices.rst +++ b/docs/topics/practices.rst @@ -101,7 +101,7 @@ reactor after ``MySpider`` has finished running. d.addBoth(lambda _: reactor.stop()) reactor.run() # the script will block here until the crawling is finished -.. seealso:: `Twisted Reactor Overview`_. +.. seealso:: :doc:`twisted:core/howto/reactor-basics` .. _run-multiple-spiders: @@ -253,6 +253,5 @@ If you are still unable to prevent your bot getting banned, consider contacting .. _ProxyMesh: https://proxymesh.com/ .. _Google cache: http://www.googleguide.com/cached_pages.html .. _testspiders: https://github.com/scrapinghub/testspiders -.. _Twisted Reactor Overview: https://twistedmatrix.com/documents/current/core/howto/reactor-basics.html .. _Crawlera: https://scrapinghub.com/crawlera .. _scrapoxy: https://scrapoxy.io/ diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index ee37f648e..4cf367d96 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -121,8 +121,8 @@ Request objects :param errback: a function that will be called if any exception was raised while processing the request. This includes pages that failed - with 404 HTTP errors and such. It receives a `Twisted Failure`_ instance - as first parameter. + with 404 HTTP errors and such. It receives a + :exc:`~twisted.python.failure.Failure` as first parameter. For more information, see :ref:`topics-request-response-ref-errbacks` below. :type errback: callable @@ -254,8 +254,8 @@ Using errbacks to catch exceptions in request processing The errback of a request is a function that will be called when an exception is raise while processing it. -It receives a `Twisted Failure`_ instance as first parameter and can be -used to track connection establishment timeouts, DNS errors etc. +It receives a :exc:`~twisted.python.failure.Failure` as first parameter and can +be used to track connection establishment timeouts, DNS errors etc. Here's an example spider logging all errors and catching some specific errors if needed:: @@ -816,5 +816,4 @@ XmlResponse objects adds encoding auto-discovering support by looking into the XML declaration line. See :attr:`TextResponse.encoding`. -.. _Twisted Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html .. _bug in lxml: https://bugs.launchpad.net/lxml/+bug/1665241 diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index ff07b9d55..3f29aa323 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -50,10 +50,10 @@ Here is a simple example showing how you can catch signals and perform some acti Deferred signal handlers ======================== -Some signals support returning `Twisted deferreds`_ from their handlers, see -the :ref:`topics-signals-ref` below to know which ones. +Some signals support returning :class:`~twisted.internet.defer.Deferred` +objects from their handlers, see the :ref:`topics-signals-ref` below to know +which ones. -.. _Twisted deferreds: https://twistedmatrix.com/documents/current/core/howto/defer.html .. _topics-signals-ref: @@ -155,8 +155,8 @@ item_error :param spider: the spider which raised the exception :type spider: :class:`~scrapy.spiders.Spider` object - :param failure: the exception raised as a Twisted `Failure`_ object - :type failure: `Failure`_ object + :param failure: the exception raised + :type failure: twisted.python.failure.Failure spider_closed ------------- @@ -236,8 +236,8 @@ spider_error This signal does not support returning deferreds from their handlers. - :param failure: the exception raised as a Twisted `Failure`_ object - :type failure: `Failure`_ object + :param failure: the exception raised + :type failure: twisted.python.failure.Failure :param response: the response being processed when the exception was raised :type response: :class:`~scrapy.http.Response` object @@ -333,5 +333,3 @@ response_downloaded :param spider: the spider for which the response is intended :type spider: :class:`~scrapy.spiders.Spider` object - -.. _Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html diff --git a/scrapy/core/downloader/contextfactory.py b/scrapy/core/downloader/contextfactory.py index 89d2776ae..6e023ebcc 100644 --- a/scrapy/core/downloader/contextfactory.py +++ b/scrapy/core/downloader/contextfactory.py @@ -67,15 +67,18 @@ class BrowserLikeContextFactory(ScrapyClientContextFactory): """ Twisted-recommended context factory for web clients. - Quoting https://twistedmatrix.com/documents/current/api/twisted.web.client.Agent.html: - "The default is to use a BrowserLikePolicyForHTTPS, - so unless you have special requirements you can leave this as-is." + Quoting the documentation of the :class:`~twisted.web.client.Agent` class: - creatorForNetloc() is the same as BrowserLikePolicyForHTTPS - except this context factory allows setting the TLS/SSL method to use. + The default is to use a + :class:`~twisted.web.client.BrowserLikePolicyForHTTPS`, so unless you + have special requirements you can leave this as-is. - Default OpenSSL method is TLS_METHOD (also called SSLv23_METHOD) - which allows TLS protocol negotiation. + :meth:`creatorForNetloc` is the same as + :class:`~twisted.web.client.BrowserLikePolicyForHTTPS` except this context + factory allows setting the TLS/SSL method to use. + + The default OpenSSL method is ``TLS_METHOD`` (also called + ``SSLv23_METHOD``) which allows TLS protocol negotiation. """ def creatorForNetloc(self, hostname, port): diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 8868a985b..f8c80880a 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -110,7 +110,7 @@ class Crawler(object): class CrawlerRunner(object): """ This is a convenient helper class that keeps track of, manages and runs - crawlers inside an already setup Twisted `reactor`_. + crawlers inside an already setup :mod:`~twisted.internet.reactor`. The CrawlerRunner object must be instantiated with a :class:`~scrapy.settings.Settings` object. @@ -233,12 +233,13 @@ class CrawlerProcess(CrawlerRunner): A class to run multiple scrapy crawlers in a process simultaneously. This class extends :class:`~scrapy.crawler.CrawlerRunner` by adding support - for starting a Twisted `reactor`_ and handling shutdown signals, like the - keyboard interrupt command Ctrl-C. It also configures top-level logging. + for starting a :mod:`~twisted.internet.reactor` and handling shutdown + signals, like the keyboard interrupt command Ctrl-C. It also configures + top-level logging. This utility should be a better fit than :class:`~scrapy.crawler.CrawlerRunner` if you aren't running another - Twisted `reactor`_ within your application. + :mod:`~twisted.internet.reactor` within your application. The CrawlerProcess object must be instantiated with a :class:`~scrapy.settings.Settings` object. @@ -273,9 +274,9 @@ class CrawlerProcess(CrawlerRunner): def start(self, stop_after_crawl=True): """ - This method starts a Twisted `reactor`_, adjusts its pool size to - :setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache based - on :setting:`DNSCACHE_ENABLED` and :setting:`DNSCACHE_SIZE`. + This method starts a :mod:`~twisted.internet.reactor`, adjusts its pool + size to :setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache + based on :setting:`DNSCACHE_ENABLED` and :setting:`DNSCACHE_SIZE`. If ``stop_after_crawl`` is True, the reactor will be stopped after all crawlers have finished, using :meth:`join`. diff --git a/scrapy/signalmanager.py b/scrapy/signalmanager.py index 296d27ed8..9a160f62e 100644 --- a/scrapy/signalmanager.py +++ b/scrapy/signalmanager.py @@ -46,16 +46,14 @@ class SignalManager(object): def send_catch_log_deferred(self, signal, **kwargs): """ - Like :meth:`send_catch_log` but supports returning `deferreds`_ from - signal handlers. + Like :meth:`send_catch_log` but supports returning + :class:`~twisted.internet.defer.Deferred` objects from signal handlers. Returns a Deferred that gets fired once all signal handlers deferreds were fired. Send a signal, catch exceptions and log them. The keyword arguments are passed to the signal handlers (connected through the :meth:`connect` method). - - .. _deferreds: https://twistedmatrix.com/documents/current/core/howto/defer.html """ kwargs.setdefault('sender', self.sender) return _signal.send_catch_log_deferred(signal, **kwargs)