mirror of https://github.com/scrapy/scrapy.git
Move scrapy/spider.py to scrapy/spiders/__init__.py
This commit is contained in:
parent
d4926091ab
commit
d3f576a816
|
|
@ -95,18 +95,18 @@ domain (or group of domains).
|
||||||
They define an initial list of URLs to download, how to follow links, and how
|
They define an initial list of URLs to download, how to follow links, and how
|
||||||
to parse the contents of pages to extract :ref:`items <topics-items>`.
|
to parse the contents of pages to extract :ref:`items <topics-items>`.
|
||||||
|
|
||||||
To create a Spider, you must subclass :class:`scrapy.Spider <scrapy.spider.Spider>` and
|
To create a Spider, you must subclass :class:`scrapy.Spider
|
||||||
define some attributes:
|
<scrapy.spiders.Spider>` and define some attributes:
|
||||||
|
|
||||||
* :attr:`~scrapy.spider.Spider.name`: identifies the Spider. It must be
|
* :attr:`~scrapy.spiders.Spider.name`: identifies the Spider. It must be
|
||||||
unique, that is, you can't set the same name for different Spiders.
|
unique, that is, you can't set the same name for different Spiders.
|
||||||
|
|
||||||
* :attr:`~scrapy.spider.Spider.start_urls`: a list of URLs where the
|
* :attr:`~scrapy.spiders.Spider.start_urls`: a list of URLs where the
|
||||||
Spider will begin to crawl from. The first pages downloaded will be those
|
Spider will begin to crawl from. The first pages downloaded will be those
|
||||||
listed here. The subsequent URLs will be generated successively from data
|
listed here. The subsequent URLs will be generated successively from data
|
||||||
contained in the start URLs.
|
contained in the start URLs.
|
||||||
|
|
||||||
* :meth:`~scrapy.spider.Spider.parse`: a method of the spider, which will
|
* :meth:`~scrapy.spiders.Spider.parse`: a method of the spider, which will
|
||||||
be called with the downloaded :class:`~scrapy.http.Response` object of each
|
be called with the downloaded :class:`~scrapy.http.Response` object of each
|
||||||
start URL. The response is passed to the method as the first and only
|
start URL. The response is passed to the method as the first and only
|
||||||
argument.
|
argument.
|
||||||
|
|
@ -114,7 +114,7 @@ define some attributes:
|
||||||
This method is responsible for parsing the response data and extracting
|
This method is responsible for parsing the response data and extracting
|
||||||
scraped data (as scraped items) and more URLs to follow.
|
scraped data (as scraped items) and more URLs to follow.
|
||||||
|
|
||||||
The :meth:`~scrapy.spider.Spider.parse` method is in charge of processing
|
The :meth:`~scrapy.spiders.Spider.parse` method is in charge of processing
|
||||||
the response and returning scraped data (as :class:`~scrapy.item.Item`
|
the response and returning scraped data (as :class:`~scrapy.item.Item`
|
||||||
objects) and more URLs to follow (as :class:`~scrapy.http.Request` objects).
|
objects) and more URLs to follow (as :class:`~scrapy.http.Request` objects).
|
||||||
|
|
||||||
|
|
@ -178,7 +178,7 @@ them the ``parse`` method of the spider as their callback function.
|
||||||
|
|
||||||
These Requests are scheduled, then executed, and :class:`scrapy.http.Response`
|
These Requests are scheduled, then executed, and :class:`scrapy.http.Response`
|
||||||
objects are returned and then fed back to the spider, through the
|
objects are returned and then fed back to the spider, through the
|
||||||
:meth:`~scrapy.spider.Spider.parse` method.
|
:meth:`~scrapy.spiders.Spider.parse` method.
|
||||||
|
|
||||||
Extracting Items
|
Extracting Items
|
||||||
----------------
|
----------------
|
||||||
|
|
|
||||||
|
|
@ -31,7 +31,7 @@ how you :ref:`configure the downloader middlewares
|
||||||
.. class:: Crawler(spidercls, settings)
|
.. class:: Crawler(spidercls, settings)
|
||||||
|
|
||||||
The Crawler object must be instantiated with a
|
The Crawler object must be instantiated with a
|
||||||
:class:`scrapy.spider.Spider` subclass and a
|
:class:`scrapy.spiders.Spider` subclass and a
|
||||||
:class:`scrapy.settings.Settings` object.
|
:class:`scrapy.settings.Settings` object.
|
||||||
|
|
||||||
.. attribute:: settings
|
.. attribute:: settings
|
||||||
|
|
|
||||||
|
|
@ -91,7 +91,7 @@ more of the following methods:
|
||||||
:type request: :class:`~scrapy.http.Request` object
|
:type request: :class:`~scrapy.http.Request` object
|
||||||
|
|
||||||
:param spider: the spider for which this request is intended
|
:param spider: the spider for which this request is intended
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. method:: process_response(request, response, spider)
|
.. method:: process_response(request, response, spider)
|
||||||
|
|
||||||
|
|
@ -118,7 +118,7 @@ more of the following methods:
|
||||||
:type response: :class:`~scrapy.http.Response` object
|
:type response: :class:`~scrapy.http.Response` object
|
||||||
|
|
||||||
:param spider: the spider for which this response is intended
|
:param spider: the spider for which this response is intended
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. method:: process_exception(request, exception, spider)
|
.. method:: process_exception(request, exception, spider)
|
||||||
|
|
||||||
|
|
@ -149,7 +149,7 @@ more of the following methods:
|
||||||
:type exception: an ``Exception`` object
|
:type exception: an ``Exception`` object
|
||||||
|
|
||||||
:param spider: the spider for which this request is intended
|
:param spider: the spider for which this request is intended
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. _topics-downloader-middleware-ref:
|
.. _topics-downloader-middleware-ref:
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -36,7 +36,7 @@ Each item pipeline component is a Python class that must implement the following
|
||||||
:type item: :class:`~scrapy.item.Item` object or a dict
|
:type item: :class:`~scrapy.item.Item` object or a dict
|
||||||
|
|
||||||
:param spider: the spider which scraped the item
|
:param spider: the spider which scraped the item
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
Additionally, they may also implement the following methods:
|
Additionally, they may also implement the following methods:
|
||||||
|
|
||||||
|
|
@ -45,14 +45,14 @@ Additionally, they may also implement the following methods:
|
||||||
This method is called when the spider is opened.
|
This method is called when the spider is opened.
|
||||||
|
|
||||||
:param spider: the spider which was opened
|
:param spider: the spider which was opened
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. method:: close_spider(self, spider)
|
.. method:: close_spider(self, spider)
|
||||||
|
|
||||||
This method is called when the spider is closed.
|
This method is called when the spider is closed.
|
||||||
|
|
||||||
:param spider: the spider which was closed
|
:param spider: the spider which was closed
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. method:: from_crawler(cls, crawler)
|
.. method:: from_crawler(cls, crawler)
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -92,7 +92,7 @@ subclasses):
|
||||||
* :class:`scrapy.http.Response`
|
* :class:`scrapy.http.Response`
|
||||||
* :class:`scrapy.item.Item`
|
* :class:`scrapy.item.Item`
|
||||||
* :class:`scrapy.selector.Selector`
|
* :class:`scrapy.selector.Selector`
|
||||||
* :class:`scrapy.spider.Spider`
|
* :class:`scrapy.spiders.Spider`
|
||||||
|
|
||||||
A real example
|
A real example
|
||||||
--------------
|
--------------
|
||||||
|
|
@ -155,7 +155,7 @@ For this reason, that function has a ``ignore`` argument which can be used to
|
||||||
ignore a particular class (and all its subclases). For
|
ignore a particular class (and all its subclases). For
|
||||||
example, this won't show any live references to spiders::
|
example, this won't show any live references to spiders::
|
||||||
|
|
||||||
>>> from scrapy.spider import Spider
|
>>> from scrapy.spiders import Spider
|
||||||
>>> prefs(ignore=Spider)
|
>>> prefs(ignore=Spider)
|
||||||
|
|
||||||
.. module:: scrapy.utils.trackref
|
.. module:: scrapy.utils.trackref
|
||||||
|
|
|
||||||
|
|
@ -94,7 +94,7 @@ path::
|
||||||
Logging from Spiders
|
Logging from Spiders
|
||||||
====================
|
====================
|
||||||
|
|
||||||
Scrapy provides a :data:`~scrapy.spider.Spider.logger` within each Spider
|
Scrapy provides a :data:`~scrapy.spiders.Spider.logger` within each Spider
|
||||||
instance, that can be accessed and used like this::
|
instance, that can be accessed and used like this::
|
||||||
|
|
||||||
import scrapy
|
import scrapy
|
||||||
|
|
|
||||||
|
|
@ -37,7 +37,7 @@ Request objects
|
||||||
request (once its downloaded) as its first parameter. For more information
|
request (once its downloaded) as its first parameter. For more information
|
||||||
see :ref:`topics-request-response-ref-request-callback-arguments` below.
|
see :ref:`topics-request-response-ref-request-callback-arguments` below.
|
||||||
If a Request doesn't specify a callback, the spider's
|
If a Request doesn't specify a callback, the spider's
|
||||||
:meth:`~scrapy.spider.Spider.parse` method will be used.
|
:meth:`~scrapy.spiders.Spider.parse` method will be used.
|
||||||
Note that if exceptions are raised during processing, errback is called instead.
|
Note that if exceptions are raised during processing, errback is called instead.
|
||||||
|
|
||||||
:type callback: callable
|
:type callback: callable
|
||||||
|
|
|
||||||
|
|
@ -67,7 +67,7 @@ Example::
|
||||||
|
|
||||||
Spiders (See the :ref:`topics-spiders` chapter for reference) can define their
|
Spiders (See the :ref:`topics-spiders` chapter for reference) can define their
|
||||||
own settings that will take precedence and override the project ones. They can
|
own settings that will take precedence and override the project ones. They can
|
||||||
do so by setting their :attr:`scrapy.spider.Spider.custom_settings` attribute.
|
do so by setting their :attr:`scrapy.spiders.Spider.custom_settings` attribute.
|
||||||
|
|
||||||
3. Project settings module
|
3. Project settings module
|
||||||
--------------------------
|
--------------------------
|
||||||
|
|
|
||||||
|
|
@ -74,7 +74,7 @@ Those objects are:
|
||||||
* ``crawler`` - the current :class:`~scrapy.crawler.Crawler` object.
|
* ``crawler`` - the current :class:`~scrapy.crawler.Crawler` object.
|
||||||
|
|
||||||
* ``spider`` - the Spider which is known to handle the URL, or a
|
* ``spider`` - the Spider which is known to handle the URL, or a
|
||||||
:class:`~scrapy.spider.Spider` object if there is no spider found for
|
:class:`~scrapy.spiders.Spider` object if there is no spider found for
|
||||||
the current URL
|
the current URL
|
||||||
|
|
||||||
* ``request`` - a :class:`~scrapy.http.Request` object of the last fetched
|
* ``request`` - a :class:`~scrapy.http.Request` object of the last fetched
|
||||||
|
|
|
||||||
|
|
@ -74,7 +74,7 @@ item_scraped
|
||||||
:type item: dict or :class:`~scrapy.item.Item` object
|
:type item: dict or :class:`~scrapy.item.Item` object
|
||||||
|
|
||||||
:param spider: the spider which scraped the item
|
:param spider: the spider which scraped the item
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
:param response: the response from where the item was scraped
|
:param response: the response from where the item was scraped
|
||||||
:type response: :class:`~scrapy.http.Response` object
|
:type response: :class:`~scrapy.http.Response` object
|
||||||
|
|
@ -94,7 +94,7 @@ item_dropped
|
||||||
:type item: dict or :class:`~scrapy.item.Item` object
|
:type item: dict or :class:`~scrapy.item.Item` object
|
||||||
|
|
||||||
:param spider: the spider which scraped the item
|
:param spider: the spider which scraped the item
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
:param response: the response from where the item was dropped
|
:param response: the response from where the item was dropped
|
||||||
:type response: :class:`~scrapy.http.Response` object
|
:type response: :class:`~scrapy.http.Response` object
|
||||||
|
|
@ -116,7 +116,7 @@ spider_closed
|
||||||
This signal supports returning deferreds from their handlers.
|
This signal supports returning deferreds from their handlers.
|
||||||
|
|
||||||
:param spider: the spider which has been closed
|
:param spider: the spider which has been closed
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
:param reason: a string which describes the reason why the spider was closed. If
|
:param reason: a string which describes the reason why the spider was closed. If
|
||||||
it was closed because the spider has completed scraping, the reason
|
it was closed because the spider has completed scraping, the reason
|
||||||
|
|
@ -140,7 +140,7 @@ spider_opened
|
||||||
This signal supports returning deferreds from their handlers.
|
This signal supports returning deferreds from their handlers.
|
||||||
|
|
||||||
:param spider: the spider which has been opened
|
:param spider: the spider which has been opened
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
spider_idle
|
spider_idle
|
||||||
-----------
|
-----------
|
||||||
|
|
@ -164,7 +164,7 @@ spider_idle
|
||||||
This signal does not support returning deferreds from their handlers.
|
This signal does not support returning deferreds from their handlers.
|
||||||
|
|
||||||
:param spider: the spider which has gone idle
|
:param spider: the spider which has gone idle
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
spider_error
|
spider_error
|
||||||
------------
|
------------
|
||||||
|
|
@ -181,7 +181,7 @@ spider_error
|
||||||
:type response: :class:`~scrapy.http.Response` object
|
:type response: :class:`~scrapy.http.Response` object
|
||||||
|
|
||||||
:param spider: the spider which raised the exception
|
:param spider: the spider which raised the exception
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
request_scheduled
|
request_scheduled
|
||||||
-----------------
|
-----------------
|
||||||
|
|
@ -198,7 +198,7 @@ request_scheduled
|
||||||
:type request: :class:`~scrapy.http.Request` object
|
:type request: :class:`~scrapy.http.Request` object
|
||||||
|
|
||||||
:param spider: the spider that yielded the request
|
:param spider: the spider that yielded the request
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
request_dropped
|
request_dropped
|
||||||
-----------------
|
-----------------
|
||||||
|
|
@ -215,7 +215,7 @@ request_dropped
|
||||||
:type request: :class:`~scrapy.http.Request` object
|
:type request: :class:`~scrapy.http.Request` object
|
||||||
|
|
||||||
:param spider: the spider that yielded the request
|
:param spider: the spider that yielded the request
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
response_received
|
response_received
|
||||||
-----------------
|
-----------------
|
||||||
|
|
@ -235,7 +235,7 @@ response_received
|
||||||
:type request: :class:`~scrapy.http.Request` object
|
:type request: :class:`~scrapy.http.Request` object
|
||||||
|
|
||||||
:param spider: the spider for which the response is intended
|
:param spider: the spider for which the response is intended
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
response_downloaded
|
response_downloaded
|
||||||
-------------------
|
-------------------
|
||||||
|
|
@ -254,6 +254,6 @@ response_downloaded
|
||||||
:type request: :class:`~scrapy.http.Request` object
|
:type request: :class:`~scrapy.http.Request` object
|
||||||
|
|
||||||
:param spider: the spider for which the response is intended
|
:param spider: the spider for which the response is intended
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. _Failure: http://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html
|
.. _Failure: http://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html
|
||||||
|
|
|
||||||
|
|
@ -81,7 +81,7 @@ following methods:
|
||||||
:type response: :class:`~scrapy.http.Response` object
|
:type response: :class:`~scrapy.http.Response` object
|
||||||
|
|
||||||
:param spider: the spider for which this response is intended
|
:param spider: the spider for which this response is intended
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
|
|
||||||
.. method:: process_spider_output(response, result, spider)
|
.. method:: process_spider_output(response, result, spider)
|
||||||
|
|
@ -102,7 +102,7 @@ following methods:
|
||||||
or :class:`~scrapy.item.Item` objects
|
or :class:`~scrapy.item.Item` objects
|
||||||
|
|
||||||
:param spider: the spider whose result is being processed
|
:param spider: the spider whose result is being processed
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
|
|
||||||
.. method:: process_spider_exception(response, exception, spider)
|
.. method:: process_spider_exception(response, exception, spider)
|
||||||
|
|
@ -130,7 +130,7 @@ following methods:
|
||||||
:type exception: `Exception`_ object
|
:type exception: `Exception`_ object
|
||||||
|
|
||||||
:param spider: the spider which raised the exception
|
:param spider: the spider which raised the exception
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
.. method:: process_start_requests(start_requests, spider)
|
.. method:: process_start_requests(start_requests, spider)
|
||||||
|
|
||||||
|
|
@ -157,7 +157,7 @@ following methods:
|
||||||
:type start_requests: an iterable of :class:`~scrapy.http.Request`
|
:type start_requests: an iterable of :class:`~scrapy.http.Request`
|
||||||
|
|
||||||
:param spider: the spider to whom the start requests belong
|
:param spider: the spider to whom the start requests belong
|
||||||
:type spider: :class:`~scrapy.spider.Spider` object
|
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||||
|
|
||||||
|
|
||||||
.. _Exception: https://docs.python.org/2/library/exceptions.html#exceptions.Exception
|
.. _Exception: https://docs.python.org/2/library/exceptions.html#exceptions.Exception
|
||||||
|
|
@ -272,7 +272,7 @@ OffsiteMiddleware
|
||||||
Filters out Requests for URLs outside the domains covered by the spider.
|
Filters out Requests for URLs outside the domains covered by the spider.
|
||||||
|
|
||||||
This middleware filters out every request whose host names aren't in the
|
This middleware filters out every request whose host names aren't in the
|
||||||
spider's :attr:`~scrapy.spider.Spider.allowed_domains` attribute.
|
spider's :attr:`~scrapy.spiders.Spider.allowed_domains` attribute.
|
||||||
|
|
||||||
When your spider returns a request for a domain not belonging to those
|
When your spider returns a request for a domain not belonging to those
|
||||||
covered by the spider, this middleware will log a debug message similar to
|
covered by the spider, this middleware will log a debug message similar to
|
||||||
|
|
@ -287,7 +287,7 @@ OffsiteMiddleware
|
||||||
will be printed (but only for the first request filtered).
|
will be printed (but only for the first request filtered).
|
||||||
|
|
||||||
If the spider doesn't define an
|
If the spider doesn't define an
|
||||||
:attr:`~scrapy.spider.Spider.allowed_domains` attribute, or the
|
:attr:`~scrapy.spiders.Spider.allowed_domains` attribute, or the
|
||||||
attribute is empty, the offsite middleware will allow all requests.
|
attribute is empty, the offsite middleware will allow all requests.
|
||||||
|
|
||||||
If the request has the :attr:`~scrapy.http.Request.dont_filter` attribute
|
If the request has the :attr:`~scrapy.http.Request.dont_filter` attribute
|
||||||
|
|
|
||||||
|
|
@ -17,10 +17,10 @@ For spiders, the scraping cycle goes through something like this:
|
||||||
those requests.
|
those requests.
|
||||||
|
|
||||||
The first requests to perform are obtained by calling the
|
The first requests to perform are obtained by calling the
|
||||||
:meth:`~scrapy.spider.Spider.start_requests` method which (by default)
|
:meth:`~scrapy.spiders.Spider.start_requests` method which (by default)
|
||||||
generates :class:`~scrapy.http.Request` for the URLs specified in the
|
generates :class:`~scrapy.http.Request` for the URLs specified in the
|
||||||
:attr:`~scrapy.spider.Spider.start_urls` and the
|
:attr:`~scrapy.spiders.Spider.start_urls` and the
|
||||||
:attr:`~scrapy.spider.Spider.parse` method as callback function for the
|
:attr:`~scrapy.spiders.Spider.parse` method as callback function for the
|
||||||
Requests.
|
Requests.
|
||||||
|
|
||||||
2. In the callback function, you parse the response (web page) and return either
|
2. In the callback function, you parse the response (web page) and return either
|
||||||
|
|
@ -42,7 +42,7 @@ Even though this cycle applies (more or less) to any kind of spider, there are
|
||||||
different kinds of default spiders bundled into Scrapy for different purposes.
|
different kinds of default spiders bundled into Scrapy for different purposes.
|
||||||
We will talk about those types here.
|
We will talk about those types here.
|
||||||
|
|
||||||
.. module:: scrapy.spider
|
.. module:: scrapy.spiders
|
||||||
:synopsis: Spiders base class, spider manager and spider middleware
|
:synopsis: Spiders base class, spider manager and spider middleware
|
||||||
|
|
||||||
.. _topics-spiders-ref:
|
.. _topics-spiders-ref:
|
||||||
|
|
@ -319,8 +319,7 @@ with a ``TestItem`` declared in a ``myproject.items`` module::
|
||||||
description = scrapy.Field()
|
description = scrapy.Field()
|
||||||
|
|
||||||
|
|
||||||
.. module:: scrapy.spiders
|
.. currentmodule:: scrapy.spiders
|
||||||
:synopsis: Collection of generic spiders
|
|
||||||
|
|
||||||
CrawlSpider
|
CrawlSpider
|
||||||
-----------
|
-----------
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@ usage:
|
||||||
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -45,7 +45,7 @@ if twisted_version >= (11, 1, 0):
|
||||||
optional_features.add('http11')
|
optional_features.add('http11')
|
||||||
|
|
||||||
# Declare top-level shortcuts
|
# Declare top-level shortcuts
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request, FormRequest
|
from scrapy.http import Request, FormRequest
|
||||||
from scrapy.selector import Selector
|
from scrapy.selector import Selector
|
||||||
from scrapy.item import Item, Field
|
from scrapy.item import Item, Field
|
||||||
|
|
|
||||||
|
|
@ -138,7 +138,7 @@ class CrawlerRunner(object):
|
||||||
:param crawler_or_spidercls: already created crawler, or a spider class
|
:param crawler_or_spidercls: already created crawler, or a spider class
|
||||||
or spider's name inside the project to create it
|
or spider's name inside the project to create it
|
||||||
:type crawler_or_spidercls: :class:`~scrapy.crawler.Crawler` instance,
|
:type crawler_or_spidercls: :class:`~scrapy.crawler.Crawler` instance,
|
||||||
:class:`~scrapy.spider.Spider` subclass or string
|
:class:`~scrapy.spiders.Spider` subclass or string
|
||||||
|
|
||||||
:param list args: arguments to initialize the spider
|
:param list args: arguments to initialize the spider
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -17,7 +17,7 @@ from scrapy.exceptions import IgnoreRequest, ScrapyDeprecationWarning
|
||||||
from scrapy.http import Request, Response
|
from scrapy.http import Request, Response
|
||||||
from scrapy.item import BaseItem
|
from scrapy.item import BaseItem
|
||||||
from scrapy.settings import Settings
|
from scrapy.settings import Settings
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.console import start_python_console
|
from scrapy.utils.console import start_python_console
|
||||||
from scrapy.utils.misc import load_object
|
from scrapy.utils.misc import load_object
|
||||||
from scrapy.utils.response import open_in_browser
|
from scrapy.utils.response import open_in_browser
|
||||||
|
|
|
||||||
|
|
@ -111,3 +111,7 @@ spiders = ObsoleteClass(
|
||||||
'it with your project settings"'
|
'it with your project settings"'
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# Top-level imports
|
||||||
|
from scrapy.spiders.crawl import CrawlSpider, Rule
|
||||||
|
from scrapy.spiders.feed import XMLFeedSpider, CSVFeedSpider
|
||||||
|
from scrapy.spiders.sitemap import SitemapSpider
|
||||||
|
|
@ -9,7 +9,7 @@ import copy
|
||||||
|
|
||||||
from scrapy.http import Request, HtmlResponse
|
from scrapy.http import Request, HtmlResponse
|
||||||
from scrapy.utils.spider import iterate_spider_output
|
from scrapy.utils.spider import iterate_spider_output
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
def identity(x):
|
def identity(x):
|
||||||
return x
|
return x
|
||||||
|
|
|
||||||
|
|
@ -4,7 +4,7 @@ for scraping from an XML feed.
|
||||||
|
|
||||||
See documentation in docs/topics/spiders.rst
|
See documentation in docs/topics/spiders.rst
|
||||||
"""
|
"""
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.iterators import xmliter, csviter
|
from scrapy.utils.iterators import xmliter, csviter
|
||||||
from scrapy.utils.spider import iterate_spider_output
|
from scrapy.utils.spider import iterate_spider_output
|
||||||
from scrapy.selector import Selector
|
from scrapy.selector import Selector
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,4 @@
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.spider import iterate_spider_output
|
from scrapy.utils.spider import iterate_spider_output
|
||||||
|
|
||||||
class InitSpider(Spider):
|
class InitSpider(Spider):
|
||||||
|
|
@ -20,8 +20,8 @@ class InitSpider(Spider):
|
||||||
is called this spider is considered initialized. If you need to perform
|
is called this spider is considered initialized. If you need to perform
|
||||||
several requests for initializing your spider, you can do so by using
|
several requests for initializing your spider, you can do so by using
|
||||||
different callbacks. The only requirement is that the final callback
|
different callbacks. The only requirement is that the final callback
|
||||||
(of the last initialization request) must be self.initialized.
|
(of the last initialization request) must be self.initialized.
|
||||||
|
|
||||||
The default implementation calls self.initialized immediately, and
|
The default implementation calls self.initialized immediately, and
|
||||||
means that no initialization is needed. This method should be
|
means that no initialization is needed. This method should be
|
||||||
overridden only when you need to perform requests to initialize your
|
overridden only when you need to perform requests to initialize your
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
import re
|
import re
|
||||||
import logging
|
import logging
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request, XmlResponse
|
from scrapy.http import Request, XmlResponse
|
||||||
from scrapy.utils.sitemap import Sitemap, sitemap_urls_from_robots
|
from scrapy.utils.sitemap import Sitemap, sitemap_urls_from_robots
|
||||||
from scrapy.utils.gz import gunzip, is_gzipped
|
from scrapy.utils.gz import gunzip, is_gzipped
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@ import inspect
|
||||||
|
|
||||||
import six
|
import six
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.misc import arg_to_iter
|
from scrapy.utils.misc import arg_to_iter
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
@ -19,7 +19,7 @@ def iter_spider_classes(module):
|
||||||
"""
|
"""
|
||||||
# this needs to be imported here until get rid of the spider manager
|
# this needs to be imported here until get rid of the spider manager
|
||||||
# singleton in scrapy.spider.spiders
|
# singleton in scrapy.spider.spiders
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
for obj in six.itervalues(vars(module)):
|
for obj in six.itervalues(vars(module)):
|
||||||
if inspect.isclass(obj) and \
|
if inspect.isclass(obj) and \
|
||||||
|
|
|
||||||
|
|
@ -27,7 +27,7 @@ def get_crawler(spidercls=None, settings_dict=None):
|
||||||
"""
|
"""
|
||||||
from scrapy.crawler import CrawlerRunner
|
from scrapy.crawler import CrawlerRunner
|
||||||
from scrapy.settings import Settings
|
from scrapy.settings import Settings
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
runner = CrawlerRunner(Settings(settings_dict))
|
runner = CrawlerRunner(Settings(settings_dict))
|
||||||
return runner._create_crawler(spidercls or Spider)
|
return runner._create_crawler(spidercls or Spider)
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,7 @@ Some spiders used for testing and benchmarking
|
||||||
import time
|
import time
|
||||||
from six.moves.urllib.parse import urlencode
|
from six.moves.urllib.parse import urlencode
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.item import Item
|
from scrapy.item import Item
|
||||||
from scrapy.linkextractors import LinkExtractor
|
from scrapy.linkextractors import LinkExtractor
|
||||||
|
|
|
||||||
|
|
@ -158,7 +158,7 @@ class MySpider(scrapy.Spider):
|
||||||
fname = abspath(join(tmpdir, 'myspider.py'))
|
fname = abspath(join(tmpdir, 'myspider.py'))
|
||||||
with open(fname, 'w') as f:
|
with open(fname, 'w') as f:
|
||||||
f.write("""
|
f.write("""
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
""")
|
""")
|
||||||
p = self.proc('runspider', fname)
|
p = self.proc('runspider', fname)
|
||||||
log = p.stderr.read()
|
log = p.stderr.read()
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from unittest import TextTestResult
|
||||||
|
|
||||||
from twisted.trial import unittest
|
from twisted.trial import unittest
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.item import Item, Field
|
from scrapy.item import Item, Field
|
||||||
from scrapy.contracts import ContractsManager
|
from scrapy.contracts import ContractsManager
|
||||||
|
|
|
||||||
|
|
@ -24,7 +24,7 @@ from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler
|
||||||
from scrapy.core.downloader.handlers.s3 import S3DownloadHandler
|
from scrapy.core.downloader.handlers.s3 import S3DownloadHandler
|
||||||
from scrapy.core.downloader.handlers.ftp import FTPDownloadHandler
|
from scrapy.core.downloader.handlers.ftp import FTPDownloadHandler
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.settings import Settings
|
from scrapy.settings import Settings
|
||||||
from scrapy import optional_features
|
from scrapy import optional_features
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from twisted.trial.unittest import TestCase
|
||||||
from twisted.python.failure import Failure
|
from twisted.python.failure import Failure
|
||||||
|
|
||||||
from scrapy.http import Request, Response
|
from scrapy.http import Request, Response
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.core.downloader.middleware import DownloaderMiddlewareManager
|
from scrapy.core.downloader.middleware import DownloaderMiddlewareManager
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.downloadermiddlewares.ajaxcrawl import AjaxCrawlMiddleware
|
from scrapy.downloadermiddlewares.ajaxcrawl import AjaxCrawlMiddleware
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request, HtmlResponse, Response
|
from scrapy.http import Request, HtmlResponse, Response
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from unittest import TestCase
|
||||||
import re
|
import re
|
||||||
|
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.downloadermiddlewares.cookies import CookiesMiddleware
|
from scrapy.downloadermiddlewares.cookies import CookiesMiddleware
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
from unittest import TestCase, main
|
from unittest import TestCase, main
|
||||||
from scrapy.http import Response, XmlResponse
|
from scrapy.http import Response, XmlResponse
|
||||||
from scrapy.downloadermiddlewares.decompression import DecompressionMiddleware
|
from scrapy.downloadermiddlewares.decompression import DecompressionMiddleware
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from tests import get_testdata
|
from tests import get_testdata
|
||||||
from scrapy.utils.test import assert_samelines
|
from scrapy.utils.test import assert_samelines
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@ import six
|
||||||
|
|
||||||
from scrapy.downloadermiddlewares.defaultheaders import DefaultHeadersMiddleware
|
from scrapy.downloadermiddlewares.defaultheaders import DefaultHeadersMiddleware
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.downloadermiddlewares.downloadtimeout import DownloadTimeoutMiddleware
|
from scrapy.downloadermiddlewares.downloadtimeout import DownloadTimeoutMiddleware
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ import unittest
|
||||||
|
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.downloadermiddlewares.httpauth import HttpAuthMiddleware
|
from scrapy.downloadermiddlewares.httpauth import HttpAuthMiddleware
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
class TestSpider(Spider):
|
class TestSpider(Spider):
|
||||||
http_user = 'foo'
|
http_user = 'foo'
|
||||||
|
|
|
||||||
|
|
@ -8,7 +8,7 @@ from contextlib import contextmanager
|
||||||
import pytest
|
import pytest
|
||||||
|
|
||||||
from scrapy.http import Response, HtmlResponse, Request
|
from scrapy.http import Response, HtmlResponse, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.settings import Settings
|
from scrapy.settings import Settings
|
||||||
from scrapy.exceptions import IgnoreRequest
|
from scrapy.exceptions import IgnoreRequest
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@ from unittest import TestCase
|
||||||
from os.path import join, abspath, dirname
|
from os.path import join, abspath, dirname
|
||||||
from gzip import GzipFile
|
from gzip import GzipFile
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Response, Request, HtmlResponse
|
from scrapy.http import Response, Request, HtmlResponse
|
||||||
from scrapy.downloadermiddlewares.httpcompression import HttpCompressionMiddleware
|
from scrapy.downloadermiddlewares.httpcompression import HttpCompressionMiddleware
|
||||||
from tests import tests_datadir
|
from tests import tests_datadir
|
||||||
|
|
|
||||||
|
|
@ -5,7 +5,7 @@ from twisted.trial.unittest import TestCase, SkipTest
|
||||||
from scrapy.downloadermiddlewares.httpproxy import HttpProxyMiddleware
|
from scrapy.downloadermiddlewares.httpproxy import HttpProxyMiddleware
|
||||||
from scrapy.exceptions import NotConfigured
|
from scrapy.exceptions import NotConfigured
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
spider = Spider('foo')
|
spider = Spider('foo')
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.downloadermiddlewares.redirect import RedirectMiddleware, MetaRefreshMiddleware
|
from scrapy.downloadermiddlewares.redirect import RedirectMiddleware, MetaRefreshMiddleware
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.exceptions import IgnoreRequest
|
from scrapy.exceptions import IgnoreRequest
|
||||||
from scrapy.http import Request, Response, HtmlResponse
|
from scrapy.http import Request, Response, HtmlResponse
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@ from twisted.internet.error import TimeoutError, DNSLookupError, \
|
||||||
from scrapy import optional_features
|
from scrapy import optional_features
|
||||||
from scrapy.downloadermiddlewares.retry import RetryMiddleware
|
from scrapy.downloadermiddlewares.retry import RetryMiddleware
|
||||||
from scrapy.xlib.tx import ResponseFailed
|
from scrapy.xlib.tx import ResponseFailed
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request, Response
|
from scrapy.http import Request, Response
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from unittest import TestCase
|
||||||
|
|
||||||
from scrapy.downloadermiddlewares.stats import DownloaderStats
|
from scrapy.downloadermiddlewares.stats import DownloaderStats
|
||||||
from scrapy.http import Request, Response
|
from scrapy.http import Request, Response
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
from unittest import TestCase
|
from unittest import TestCase
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.downloadermiddlewares.useragent import UserAgentMiddleware
|
from scrapy.downloadermiddlewares.useragent import UserAgentMiddleware
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
|
||||||
|
|
@ -22,7 +22,7 @@ from scrapy import signals
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
from scrapy.xlib.pydispatch import dispatcher
|
from scrapy.xlib.pydispatch import dispatcher
|
||||||
from tests import tests_datadir
|
from tests import tests_datadir
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.item import Item, Field
|
from scrapy.item import Item, Field
|
||||||
from scrapy.linkextractors import LinkExtractor
|
from scrapy.linkextractors import LinkExtractor
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.http import Request, Response
|
from scrapy.http import Request, Response
|
||||||
from scrapy.item import Item, Field
|
from scrapy.item import Item, Field
|
||||||
from scrapy.logformatter import LogFormatter
|
from scrapy.logformatter import LogFormatter
|
||||||
|
|
|
||||||
|
|
@ -6,7 +6,7 @@ from twisted.internet import reactor
|
||||||
from twisted.internet.defer import Deferred, inlineCallbacks
|
from twisted.internet.defer import Deferred, inlineCallbacks
|
||||||
|
|
||||||
from scrapy.http import Request, Response
|
from scrapy.http import Request, Response
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.request import request_fingerprint
|
from scrapy.utils.request import request_fingerprint
|
||||||
from scrapy.pipelines.media import MediaPipeline
|
from scrapy.pipelines.media import MediaPipeline
|
||||||
from scrapy.utils.signal import disconnect_all
|
from scrapy.utils.signal import disconnect_all
|
||||||
|
|
|
||||||
|
|
@ -7,11 +7,10 @@ from testfixtures import LogCapture
|
||||||
from twisted.trial import unittest
|
from twisted.trial import unittest
|
||||||
|
|
||||||
from scrapy import signals
|
from scrapy import signals
|
||||||
from scrapy.spider import Spider, BaseSpider
|
|
||||||
from scrapy.settings import Settings
|
from scrapy.settings import Settings
|
||||||
from scrapy.http import Request, Response, TextResponse, XmlResponse, HtmlResponse
|
from scrapy.http import Request, Response, TextResponse, XmlResponse, HtmlResponse
|
||||||
from scrapy.spiders.init import InitSpider
|
from scrapy.spiders.init import InitSpider
|
||||||
from scrapy.spiders import CrawlSpider, Rule, XMLFeedSpider, \
|
from scrapy.spiders import Spider, BaseSpider, CrawlSpider, Rule, XMLFeedSpider, \
|
||||||
CSVFeedSpider, SitemapSpider
|
CSVFeedSpider, SitemapSpider
|
||||||
from scrapy.linkextractors import LinkExtractor
|
from scrapy.linkextractors import LinkExtractor
|
||||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||||
|
|
@ -116,7 +115,7 @@ class SpiderTest(unittest.TestCase):
|
||||||
|
|
||||||
def test_log(self):
|
def test_log(self):
|
||||||
spider = self.spider_class('example.com')
|
spider = self.spider_class('example.com')
|
||||||
with mock.patch('scrapy.spider.Spider.logger') as mock_logger:
|
with mock.patch('scrapy.spiders.Spider.logger') as mock_logger:
|
||||||
spider.log('test log msg', 'INFO')
|
spider.log('test log msg', 'INFO')
|
||||||
mock_logger.log.assert_called_once_with('INFO', 'test log msg')
|
mock_logger.log.assert_called_once_with('INFO', 'test log msg')
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -6,7 +6,7 @@ from zope.interface.verify import verifyObject
|
||||||
from twisted.trial import unittest
|
from twisted.trial import unittest
|
||||||
|
|
||||||
|
|
||||||
# ugly hack to avoid cyclic imports of scrapy.spider when running this test
|
# ugly hack to avoid cyclic imports of scrapy.spiders when running this test
|
||||||
# alone
|
# alone
|
||||||
from scrapy.interfaces import ISpiderLoader
|
from scrapy.interfaces import ISpiderLoader
|
||||||
from scrapy.spiderloader import SpiderLoader
|
from scrapy.spiderloader import SpiderLoader
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,4 @@
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
class Spider0(Spider):
|
class Spider0(Spider):
|
||||||
allowed_domains = ["scrapy1.org", "scrapy3.org"]
|
allowed_domains = ["scrapy1.org", "scrapy3.org"]
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,4 @@
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
class Spider1(Spider):
|
class Spider1(Spider):
|
||||||
name = "spider1"
|
name = "spider1"
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,4 @@
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
class Spider2(Spider):
|
class Spider2(Spider):
|
||||||
name = "spider2"
|
name = "spider2"
|
||||||
|
|
|
||||||
|
|
@ -1,4 +1,4 @@
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
class Spider3(Spider):
|
class Spider3(Spider):
|
||||||
name = "spider3"
|
name = "spider3"
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from unittest import TestCase
|
||||||
|
|
||||||
from scrapy.spidermiddlewares.depth import DepthMiddleware
|
from scrapy.spidermiddlewares.depth import DepthMiddleware
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.statscollectors import StatsCollector
|
from scrapy.statscollectors import StatsCollector
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -7,7 +7,7 @@ from twisted.internet import defer
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
from tests.mockserver import MockServer
|
from tests.mockserver import MockServer
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.spidermiddlewares.httperror import HttpErrorMiddleware, HttpError
|
from scrapy.spidermiddlewares.httperror import HttpErrorMiddleware, HttpError
|
||||||
from scrapy.settings import Settings
|
from scrapy.settings import Settings
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@ from unittest import TestCase
|
||||||
from six.moves.urllib.parse import urlparse
|
from six.moves.urllib.parse import urlparse
|
||||||
|
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.spidermiddlewares.offsite import OffsiteMiddleware
|
from scrapy.spidermiddlewares.offsite import OffsiteMiddleware
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
from unittest import TestCase
|
from unittest import TestCase
|
||||||
|
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.spidermiddlewares.referer import RefererMiddleware
|
from scrapy.spidermiddlewares.referer import RefererMiddleware
|
||||||
|
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -2,7 +2,7 @@ from unittest import TestCase
|
||||||
|
|
||||||
from scrapy.spidermiddlewares.urllength import UrlLengthMiddleware
|
from scrapy.spidermiddlewares.urllength import UrlLengthMiddleware
|
||||||
from scrapy.http import Response, Request
|
from scrapy.http import Response, Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
|
|
||||||
class TestUrlLengthMiddleware(TestCase):
|
class TestUrlLengthMiddleware(TestCase):
|
||||||
|
|
|
||||||
|
|
@ -3,7 +3,7 @@ from datetime import datetime
|
||||||
from twisted.trial import unittest
|
from twisted.trial import unittest
|
||||||
|
|
||||||
from scrapy.extensions.spiderstate import SpiderState
|
from scrapy.extensions.spiderstate import SpiderState
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
|
|
||||||
|
|
||||||
class SpiderStateTest(unittest.TestCase):
|
class SpiderStateTest(unittest.TestCase):
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.statscollectors import StatsCollector, DummyStatsCollector
|
from scrapy.statscollectors import StatsCollector, DummyStatsCollector
|
||||||
from scrapy.utils.test import get_crawler
|
from scrapy.utils.test import get_crawler
|
||||||
|
|
||||||
|
|
|
||||||
|
|
@ -21,7 +21,7 @@ class ToplevelTestCase(TestCase):
|
||||||
self.assertIs(scrapy.FormRequest, FormRequest)
|
self.assertIs(scrapy.FormRequest, FormRequest)
|
||||||
|
|
||||||
def test_spider_shortcut(self):
|
def test_spider_shortcut(self):
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
self.assertIs(scrapy.Spider, Spider)
|
self.assertIs(scrapy.Spider, Spider)
|
||||||
|
|
||||||
def test_selector_shortcut(self):
|
def test_selector_shortcut(self):
|
||||||
|
|
|
||||||
|
|
@ -1,7 +1,7 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.http import Request
|
from scrapy.http import Request
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.reqser import request_to_dict, request_from_dict
|
from scrapy.utils.reqser import request_to_dict, request_from_dict
|
||||||
|
|
||||||
class RequestSerializationTest(unittest.TestCase):
|
class RequestSerializationTest(unittest.TestCase):
|
||||||
|
|
|
||||||
|
|
@ -1,6 +1,6 @@
|
||||||
import unittest
|
import unittest
|
||||||
|
|
||||||
from scrapy.spider import Spider
|
from scrapy.spiders import Spider
|
||||||
from scrapy.utils.url import url_is_from_any_domain, url_is_from_spider, canonicalize_url
|
from scrapy.utils.url import url_is_from_any_domain, url_is_from_spider, canonicalize_url
|
||||||
|
|
||||||
__doctests__ = ['scrapy.utils.url']
|
__doctests__ = ['scrapy.utils.url']
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue