diff --git a/.gitignore b/.gitignore index 406146e5f..ff6e2ea65 100644 --- a/.gitignore +++ b/.gitignore @@ -12,6 +12,7 @@ dist .idea htmlcov/ .coverage +.pytest_cache/ .coverage.* .cache/ diff --git a/docs/index.rst b/docs/index.rst index 7e8c979c4..0a96aa88e 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -151,8 +151,7 @@ Solving specific problems topics/contracts topics/practices topics/broad-crawls - topics/firefox - topics/firebug + topics/developer-tools topics/leaks topics/media-pipeline topics/deploy @@ -175,11 +174,8 @@ Solving specific problems :doc:`topics/broad-crawls` Tune Scrapy for crawling a lot domains in parallel. -:doc:`topics/firefox` - Learn how to scrape with Firefox and some useful add-ons. - -:doc:`topics/firebug` - Learn how to scrape efficiently using Firebug. +:doc:`topics/developer-tools` + Learn how to scrape with your browser's developer tools. :doc:`topics/leaks` Learn how to find and get rid of memory leaks in your crawler. diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 0db6a6218..fa6dc274d 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -298,8 +298,7 @@ expressions`:: In order to find the proper CSS selectors to use, you might find useful opening the response page from the shell in your web browser using ``view(response)``. -You can use your browser developer tools or extensions like Firebug (see -sections about :ref:`topics-firebug` and :ref:`topics-firefox`). +You can use your browser developer tools (see section about :ref:`topics-developer-tools`). `Selector Gadget`_ is also a nice tool to quickly find CSS selector for visually selected elements, which works in many browsers. diff --git a/docs/topics/_images/firebug1.png b/docs/topics/_images/firebug1.png deleted file mode 100644 index e2eaefa83..000000000 Binary files a/docs/topics/_images/firebug1.png and /dev/null differ diff --git a/docs/topics/_images/firebug2.png b/docs/topics/_images/firebug2.png deleted file mode 100644 index 4cab63431..000000000 Binary files a/docs/topics/_images/firebug2.png and /dev/null differ diff --git a/docs/topics/_images/firebug3.png b/docs/topics/_images/firebug3.png deleted file mode 100644 index affbe14bc..000000000 Binary files a/docs/topics/_images/firebug3.png and /dev/null differ diff --git a/docs/topics/_images/inspector_01.png b/docs/topics/_images/inspector_01.png new file mode 100644 index 000000000..edb8795dc Binary files /dev/null and b/docs/topics/_images/inspector_01.png differ diff --git a/docs/topics/_images/network_01.png b/docs/topics/_images/network_01.png new file mode 100644 index 000000000..1788ea76a Binary files /dev/null and b/docs/topics/_images/network_01.png differ diff --git a/docs/topics/_images/network_02.png b/docs/topics/_images/network_02.png new file mode 100644 index 000000000..5d39ae601 Binary files /dev/null and b/docs/topics/_images/network_02.png differ diff --git a/docs/topics/_images/network_03.png b/docs/topics/_images/network_03.png new file mode 100644 index 000000000..472fca958 Binary files /dev/null and b/docs/topics/_images/network_03.png differ diff --git a/docs/topics/contracts.rst b/docs/topics/contracts.rst index ba1421c42..70f20d4ed 100644 --- a/docs/topics/contracts.rst +++ b/docs/topics/contracts.rst @@ -86,8 +86,11 @@ override three methods: .. method:: Contract.adjust_request_args(args) This receives a ``dict`` as an argument containing default arguments - for :class:`~scrapy.http.Request` object. Must return the same or a - modified version of it. + for request object. :class:`~scrapy.http.Request` is used by default, + but this can be changed with the ``request_cls`` attribute. + If multiple contracts in chain have this attribute defined, the last one is used. + + Must return the same or a modified version of it. .. method:: Contract.pre_process(response) diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst new file mode 100644 index 000000000..c1976258d --- /dev/null +++ b/docs/topics/developer-tools.rst @@ -0,0 +1,268 @@ +.. _topics-developer-tools: + +================================================= +Using your browser's Developer Tools for scraping +================================================= + +Here is a general guide on how to use your browser's Developer Tools +to ease the scraping process. Today almost all browsers come with +built in `Developer Tools`_ and although we will use Firefox in this +guide, the concepts are applicable to any other browser. + +In this guide we'll introduce the basic tools to use from a browser's +Developer Tools by scraping `quotes.toscrape.com`_. + +.. _topics-livedom: + +Caveats with inspecting the live browser DOM +============================================ + +Since Developer Tools operate on a live browser DOM, what you'll actually see +when inspecting the page source is not the original HTML, but a modified one +after applying some browser clean up and executing Javascript code. Firefox, +in particular, is known for adding ```` elements to tables. Scrapy, on +the other hand, does not modify the original page HTML, so you won't be able to +extract any data if you use ```` in your XPath expressions. + +Therefore, you should keep in mind the following things: + +* Disable Javascript while inspecting the DOM looking for XPaths to be + used in Scrapy (in the Developer Tools settings click `Disable JavaScript`) + +* Never use full XPath paths, use relative and clever ones based on attributes + (such as ``id``, ``class``, ``width``, etc) or any identifying features like + ``contains(@href, 'image')``. + +* Never include ```` elements in your XPath expressions unless you + really know what you're doing + +.. _topics-inspector: + +Inspecting a website +=================================== + +By far the most handy feature of the Developer Tools is the `Inspector` +feature, which allows you to inspect the underlying HTML code of +any webpage. To demonstrate the Inspector, let's look at the +`quotes.toscrape.com`_-site. + +On the site we have a total of ten quotes from various authors with specific +tags, as well as the Top Ten Tags. Let's say we want to extract all the quotes +on this page, without any meta-information about authors, tags, etc. + +Instead of viewing the whole source code for the page, we can simply right click +on a quote and select ``Inspect Element (Q)``, which opens up the `Inspector`. +In it you should see something like this: + +.. image:: _images/inspector_01.png + :width: 777 + :height: 469 + :alt: Firefox's Inspector-tool + +The interesting part for us is this: + +.. code-block:: html + +
+ (...) + (...) +
(...)
+
+ +If you hover over the first ``div`` directly above the ``span`` tag highlighted +in the screenshot, you'll see that the corresponding section of the webpage gets +highlighted as well. So now we have a section, but we can't find our quote text +anywhere. + +The advantage of the `Inspector` is that it automatically expands and collapses +sections and tags of a webpage, which greatly improves readability. You can +expand and collapse a tag by clicking on the arrow in front of it or by double +clicking directly on the tag. If we expand the ``span`` tag with the ``class= +"text"`` we will see the quote-text we clicked on. The `Inspector` lets you +copy XPaths to selected elements. Let's try it out: Right-click on the ``span`` +tag, select ``Copy > XPath`` and paste it in the scrapy shell like so:: + + $ scrapy shell "http://quotes.toscrape.com/" + (...) + >>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall() + ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”] + +Adding ``text()`` at the end we are able to extract the first quote with this +basic selector. But this XPath is not really that clever. All it does is +go down a desired path in the source code starting from ``html``. So let's +see if we can refine our XPath a bit: + +If we check the `Inspector` again we'll see that directly beneath our +expanded ``div`` tag we have nine identical ``div`` tags, each with the +same attributes as our first. If we expand any of them, we'll see the same +structure as with our first quote: Two ``span`` tags and one ``div`` tag. We can +expand each ``span`` tag with the ``class="text"`` inside our ``div`` tags and +see each quote: + +.. code-block:: html + +
+ + “The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.” + + (...) +
(...)
+
+ + +With this knowledge we can refine our XPath: Instead of a path to follow, +we'll simply select all ``span`` tags with the ``class="text"`` by using +the `has-class-extension`_:: + + >>> response.xpath('//span[has-class("text")]/text()').getall() + ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”, + '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', + '“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', + (...)] + +And with one simple, cleverer XPath we are able to extract all quotes from +the page. We could have constructed a loop over our first XPath to increase +the number of the last ``div``, but this would have been unnecessarily +complex and by simply constructing an XPath with ``has-class("text")`` +we were able to extract all quotes in one line. + +The `Inspector` has a lot of other helpful features, such as searching in the +source code or directly scrolling to an element you selected. Let's demonstrate +a use case: + +Say you want to find the ``Next`` button on the page. Type ``Next`` into the +search bar on the top right of the `Inspector`. You should get two results. +The first is a ``li`` tag with the ``class="text"``, the second the text +of an ``a`` tag. Right click on the ``a`` tag and select ``Scroll into View``. +If you hover over the tag, you'll see the button highlighted. From here +we could easily create a :ref:`Link Extractor ` to +follow the pagination. On a simple site such as this, there may not be +the need to find an element visually but the ``Scroll into View`` function +can be quite useful on complex sites. + +Note that the search bar can also be used to search for and test CSS +selectors. For example, you could search for ``span.text`` to find +all quote texts. Instead of a full text search, this searches for +exactly the ``span`` tag with the ``class="text"`` in the page. + +.. _topics-network-tool: + +The Network-tool +================ +While scraping you may come across dynamic webpages where some parts +of the page are loaded dynamically through multiple requests. While +this can be quite tricky, the `Network`-tool in the Developer Tools +greatly facilitates this task. To demonstrate the Network-tool, let's +take a look at the page `quotes.toscrape.com/scroll`_. + +The page is quite similar to the basic `quotes.toscrape.com`_-page, +but instead of the above-mentioned ``Next`` button, the page +automatically loads new quotes when you scroll to the bottom. We +could go ahead and try out different XPaths directly, but instead +we'll check another quite useful command from the scrapy shell:: + + $ scrapy shell "quotes.toscrape.com/scroll" + (...) + >>> view(response) + +A browser window should open with the webpage but with one +crucial difference: Instead of the quotes we just see a greenish +bar with the word ``Loading...``. + +.. image:: _images/network_01.png + :width: 777 + :height: 296 + :alt: Response from quotes.toscrape.com/scroll + +The ``view(response)`` command let's us view the response our +shell or later our spider receives from the server. Here we see +that some basic template is loaded which includes the title, +the login-button and the footer, but the quotes are missing. This +tells us that the quotes are being loaded from a different request +than ``quotes.toscrape/scroll``. + +If you click on the ``Network`` tab, you will probably only see +two entries. The first thing we do is enable persistent logs by +clicking on ``Persist Logs``. If this option is disabled, the +log is automatically cleared each time you navigate to a different +page. Enabling this option is a good default, since it gives us +control on when to clear the logs. + +If we reload the page now, you'll see the log get populated with six +new requests. + +.. image:: _images/network_02.png + :width: 777 + :height: 241 + :alt: Network tab with persistent logs and requests + +Here we see every request that has been made when reloading the page +and can inspect each request and its response. So let's find out +where our quotes are coming from: + +First click on the request with the name ``scroll``. On the right +you can now inspect the request. In ``Headers`` you'll find details +about the request headers, such as the URL, the method, the IP-address, +and so on. We'll ignore the other tabs and click directly on ``Reponse``. + +What you should see in the ``Preview`` pane is the rendered HTML-code, +that is exactly what we saw when we called ``view(response)`` in the +shell. Accordingly the ``type`` of the request in the log is ``html``. +The other requests have types like ``css`` or ``js``, but what +interests us is the one request called ``quotes?page=1`` with the +type ``json``. + +If we click on this request, we see that the request URL is +``http://quotes.toscrape.com/api/quotes?page=1`` and the response +is a JSON-object that contains our quotes. We can also right-click +on the request and open ``Open in new tab`` to get a better overview. + +.. image:: _images/network_03.png + :width: 777 + :height: 375 + :alt: JSON-object returned from the quotes.toscrape API + +With this response we can now easily parse the JSON-object and +also request each page to get every quote on the site:: + + import scrapy + import json + + + class QuoteSpider(scrapy.Spider): + name = 'quote' + allowed_domains = ['quotes.toscrape.com'] + page = 1 + start_urls = ['http://quotes.toscrape.com/api/quotes?page=1] + + def parse(self, response): + data = json.loads(response.text) + for quote in data["quotes"]: + yield {"quote": quote["text"]} + if data["has_next"]: + self.page += 1 + url = "http://quotes.toscrape.com/api/quotes?page={}".format(self.page) + yield scrapy.Request(url=url, callback=self.parse) + +This spider starts at the first page of the quotes-API. With each +response, we parse the ``response.text`` and assign it to ``data``. +This lets us operate on the JSON-object like on a Python dictionary. +We iterate through the ``quotes`` and print out the ``quote["text"]``. +If the handy ``has_next`` element is ``true`` (try loading +`quotes.toscrape.com/api/quotes?page=10`_ in your browser or a +page-number greater than 10), we increment the ``page`` attribute +and ``yield`` a new request, inserting the incremented page-number +into our ``url``. + +You can see that with a few inspections in the `Network`-tool we +were able to easily replicate the dynamic requests of the scrolling +functionality of the page. Crawling dynamic pages can be quite +daunting and pages can be very complex, but it (mostly) boils down +to identifying the correct request and replicating it in your spider. + +.. _Developer Tools: https://en.wikipedia.org/wiki/Web_development_tools +.. _quotes.toscrape.com: http://quotes.toscrape.com +.. _quotes.toscrape.com/scroll: quotes.toscrape.com/scroll/ +.. _quotes.toscrape.com/api/quotes?page=10: http://quotes.toscrape.com/api/quotes?page=10 +.. _has-class-extension: https://parsel.readthedocs.io/en/latest/usage.html#other-xpath-extensions + diff --git a/docs/topics/firebug.rst b/docs/topics/firebug.rst deleted file mode 100644 index 4ea8d3bd0..000000000 --- a/docs/topics/firebug.rst +++ /dev/null @@ -1,167 +0,0 @@ -.. _topics-firebug: - -========================== -Using Firebug for scraping -========================== - -.. note:: Google Directory, the example website used in this guide is no longer - available as it `has been shut down by Google`_. The concepts in this guide - are still valid though. If you want to update this guide to use a new - (working) site, your contribution will be more than welcome!. See :ref:`topics-contributing` - for information on how to do so. - -Introduction -============ - -This document explains how to use `Firebug`_ (a Firefox add-on) to make the -scraping process easier and more fun. For other useful Firefox add-ons see -:ref:`topics-firefox-addons`. There are some caveats with using Firefox add-ons -to inspect pages, see :ref:`topics-firefox-livedom`. - -In this example, we'll show how to use `Firebug`_ to scrape data from the -`Google Directory`_, which contains the same data as the `Open Directory -Project`_ used in the :ref:`tutorial ` but with a different -face. - -.. _Firebug: https://getfirebug.com/ -.. _Google Directory: http://directory.google.com/ -.. _Open Directory Project: http://www.dmoz.org - -Firebug comes with a very useful feature called `Inspect Element`_ which allows -you to inspect the HTML code of the different page elements just by hovering -your mouse over them. Otherwise you would have to search for the tags manually -through the HTML body which can be a very tedious task. - -.. _Inspect Element: https://www.youtube.com/watch?v=-pT_pDe54aA - -In the following screenshot you can see the `Inspect Element`_ tool in action. - -.. image:: _images/firebug1.png - :width: 913 - :height: 600 - :alt: Inspecting elements with Firebug - -At first sight, we can see that the directory is divided in categories, which -are also divided in subcategories. - -However, it seems that there are more subcategories than the ones being shown -in this page, so we'll keep looking: - -.. image:: _images/firebug2.png - :width: 819 - :height: 629 - :alt: Inspecting elements with Firebug - -As expected, the subcategories contain links to other subcategories, and also -links to actual websites, which is the purpose of the directory. - -Getting links to follow -======================= - -By looking at the category URLs we can see they share a pattern: - - http://directory.google.com/Category/Subcategory/Another_Subcategory - -Once we know that, we are able to construct a regular expression to follow -those links. For example, the following one:: - - directory\.google\.com/[A-Z][a-zA-Z_/]+$ - -So, based on that regular expression we can create the first crawling rule:: - - Rule(LinkExtractor(allow='directory.google.com/[A-Z][a-zA-Z_/]+$', ), - 'parse_category', - follow=True, - ), - -The :class:`~scrapy.spiders.Rule` object instructs -:class:`~scrapy.spiders.CrawlSpider` based spiders how to follow the -category links. ``parse_category`` will be a method of the spider which will -process and extract data from those pages. - -This is how the spider would look so far:: - - from scrapy.linkextractors import LinkExtractor - from scrapy.spiders import CrawlSpider, Rule - - class GoogleDirectorySpider(CrawlSpider): - name = 'directory.google.com' - allowed_domains = ['directory.google.com'] - start_urls = ['http://directory.google.com/'] - - rules = ( - Rule(LinkExtractor(allow='directory\.google\.com/[A-Z][a-zA-Z_/]+$'), - 'parse_category', follow=True, - ), - ) - - def parse_category(self, response): - # write the category page data extraction code here - pass - - -Extracting the data -=================== - -Now we're going to write the code to extract data from those pages. - -With the help of Firebug, we'll take a look at some page containing links to -websites (say http://directory.google.com/Top/Arts/Awards/) and find out how we can -extract those links using :ref:`Selectors `. We'll also -use the :ref:`Scrapy shell ` to test those XPath's and make sure -they work as we expect. - -.. image:: _images/firebug3.png - :width: 965 - :height: 751 - :alt: Inspecting elements with Firebug - -As you can see, the page markup is not very descriptive: the elements don't -contain ``id``, ``class`` or any attribute that clearly identifies them, so -we'll use the ranking bars as a reference point to select the data to extract -when we construct our XPaths. - -After using FireBug, we can see that each link is inside a ``td`` tag, which is -itself inside a ``tr`` tag that also contains the link's ranking bar (in -another ``td``). - -So we can select the ranking bar, then find its parent (the ``tr``), and then -finally, the link's ``td`` (which contains the data we want to scrape). - -This results in the following XPath:: - - //td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td//a - -It's important to use the :ref:`Scrapy shell ` to test these -complex XPath expressions and make sure they work as expected. - -Basically, that expression will look for the ranking bar's ``td`` element, and -then select any ``td`` element who has a descendant ``a`` element whose -``href`` attribute contains the string ``#pagerank``" - -Of course, this is not the only XPath, and maybe not the simpler one to select -that data. Another approach could be, for example, to find any ``font`` tags -that have that grey colour of the links, - -Finally, we can write our ``parse_category()`` method:: - - def parse_category(self, response): - # The path to website links in directory page - links = response.xpath('//td[descendant::a[contains(@href, "#pagerank")]]/following-sibling::td/font') - - for link in links: - item = DirectoryItem() - item['name'] = link.xpath('a/text()').extract() - item['url'] = link.xpath('a/@href').extract() - item['description'] = link.xpath('font[2]/text()').extract() - yield item - - -Be aware that you may find some elements which appear in Firebug but -not in the original HTML, such as the typical case of ```` -elements. - -or tags which Therefer in page HTML -sources may on Firebug inspects the live DOM - -.. _has been shut down by Google: https://searchenginewatch.com/sew/news/2096661/google-directory-shut diff --git a/docs/topics/firefox.rst b/docs/topics/firefox.rst deleted file mode 100644 index 2c85848be..000000000 --- a/docs/topics/firefox.rst +++ /dev/null @@ -1,82 +0,0 @@ -.. _topics-firefox: - -========================== -Using Firefox for scraping -========================== - -Here is a list of tips and advice on using Firefox for scraping, along with a -list of useful Firefox add-ons to ease the scraping process. - -.. _topics-firefox-livedom: - -Caveats with inspecting the live browser DOM -============================================ - -Since Firefox add-ons operate on a live browser DOM, what you'll actually see -when inspecting the page source is not the original HTML, but a modified one -after applying some browser clean up and executing Javascript code. Firefox, -in particular, is known for adding ```` elements to tables. Scrapy, on -the other hand, does not modify the original page HTML, so you won't be able to -extract any data if you use ```` in your XPath expressions. - -Therefore, you should keep in mind the following things when working with -Firefox and XPath: - -* Disable Firefox Javascript while inspecting the DOM looking for XPaths to be - used in Scrapy - -* Never use full XPath paths, use relative and clever ones based on attributes - (such as ``id``, ``class``, ``width``, etc) or any identifying features like - ``contains(@href, 'image')``. - -* Never include ```` elements in your XPath expressions unless you - really know what you're doing - -.. _topics-firefox-addons: - -Useful Firefox add-ons for scraping -=================================== - -Firebug -------- - -`Firebug`_ is a widely known tool among web developers and it's also very -useful for scraping. In particular, its `Inspect Element`_ feature comes very -handy when you need to construct the XPaths for extracting data because it -allows you to view the HTML code of each page element while moving your mouse -over it. - -See :ref:`topics-firebug` for a detailed guide on how to use Firebug with -Scrapy. - -XPather -------- - -`XPather`_ allows you to test XPath expressions directly on the pages. - -XPath Checker -------------- - -`XPath Checker`_ is another Firefox add-on for testing XPaths on your pages. - -Tamper Data ------------ - -`Tamper Data`_ is a Firefox add-on which allows you to view and modify the HTTP -request headers sent by Firefox. Firebug also allows to view HTTP headers, but -not to modify them. - -Firecookie ----------- - -`Firecookie`_ makes it easier to view and manage cookies. You can use this -extension to create a new cookie, delete existing cookies, see a list of cookies -for the current site, manage cookies permissions and a lot more. - -.. _Firebug: https://getfirebug.com/ -.. _Inspect Element: https://www.youtube.com/watch?v=-pT_pDe54aA -.. _XPather: https://addons.mozilla.org/en-US/firefox/addon/xpather/ -.. _XPath Checker: https://addons.mozilla.org/en-US/firefox/addon/xpath-checker/ -.. _Tamper Data: https://addons.mozilla.org/en-US/firefox/addon/tamper-data/ -.. _Firecookie: https://addons.mozilla.org/en-US/firefox/addon/firecookie/ - diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index d40c0e1df..ff07b9d55 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -279,6 +279,22 @@ request_dropped :param spider: the spider that yielded the request :type spider: :class:`~scrapy.spiders.Spider` object +request_reached_downloader +--------------------------- + +.. signal:: request_reached_downloader +.. function:: request_reached_downloader(request, spider) + + Sent when a :class:`~scrapy.http.Request` reached downloader. + + The signal does not support returning deferreds from their handlers. + + :param request: the request that reached downloader + :type request: :class:`~scrapy.http.Request` object + + :param spider: the spider that yielded the request + :type spider: :class:`~scrapy.spiders.Spider` object + response_received ----------------- diff --git a/scrapy/contracts/__init__.py b/scrapy/contracts/__init__.py index 518224225..f605c9eb1 100644 --- a/scrapy/contracts/__init__.py +++ b/scrapy/contracts/__init__.py @@ -49,8 +49,13 @@ class ContractsManager(object): def from_method(self, method, results): contracts = self.extract_contracts(method) if contracts: + request_cls = Request + for contract in contracts: + if contract.request_cls is not None: + request_cls = contract.request_cls + # calculate request args - args, kwargs = get_spec(Request.__init__) + args, kwargs = get_spec(request_cls.__init__) # Don't filter requests to allow # testing different callbacks on the same URL. @@ -60,10 +65,11 @@ class ContractsManager(object): for contract in contracts: kwargs = contract.adjust_request_args(kwargs) - # create and prepare request args.remove('self') + + # check if all positional arguments are defined in kwargs if set(args).issubset(set(kwargs)): - request = Request(**kwargs) + request = request_cls(**kwargs) # execute pre and post hooks in order for contract in reversed(contracts): @@ -99,6 +105,7 @@ class ContractsManager(object): class Contract(object): """ Abstract class for contracts """ + request_cls = None def __init__(self, method, *args): self.testcase_pre = _create_testcase(method, '@%s pre-hook' % self.name) diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index d835e65f7..59c3ad074 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -129,6 +129,9 @@ class Downloader(object): return response slot.active.add(request) + self.signals.send_catch_log(signal=signals.request_reached_downloader, + request=request, + spider=spider) deferred = defer.Deferred().addBoth(_deactivate) slot.queue.append((request, deferred)) self._process_queue(spider, slot) diff --git a/scrapy/signals.py b/scrapy/signals.py index e36c27203..c0e4bb74e 100644 --- a/scrapy/signals.py +++ b/scrapy/signals.py @@ -13,6 +13,7 @@ spider_closed = object() spider_error = object() request_scheduled = object() request_dropped = object() +request_reached_downloader = object() response_received = object() response_downloaded = object() item_scraped = object() diff --git a/scrapy/spidermiddlewares/offsite.py b/scrapy/spidermiddlewares/offsite.py index 310166cad..232e96cbb 100644 --- a/scrapy/spidermiddlewares/offsite.py +++ b/scrapy/spidermiddlewares/offsite.py @@ -3,7 +3,6 @@ Offsite Spider Middleware See documentation in docs/topics/spider-middleware.rst """ - import re import logging import warnings @@ -35,8 +34,9 @@ class OffsiteMiddleware(object): domain = urlparse_cached(x).hostname if domain and domain not in self.domains_seen: self.domains_seen.add(domain) - logger.debug("Filtered offsite request to %(domain)r: %(request)s", - {'domain': domain, 'request': x}, extra={'spider': spider}) + logger.debug( + "Filtered offsite request to %(domain)r: %(request)s", + {'domain': domain, 'request': x}, extra={'spider': spider}) self.stats.inc_value('offsite/domains', spider=spider) self.stats.inc_value('offsite/filtered', spider=spider) else: @@ -52,13 +52,15 @@ class OffsiteMiddleware(object): """Override this method to implement a different offsite policy""" allowed_domains = getattr(spider, 'allowed_domains', None) if not allowed_domains: - return re.compile('') # allow all by default + return re.compile('') # allow all by default url_pattern = re.compile("^https?://.*$") for domain in allowed_domains: if url_pattern.match(domain): - warnings.warn("allowed_domains accepts only domains, not URLs. Ignoring URL entry %s in allowed_domains." % domain, URLWarning) - - regex = r'^(.*\.)?(%s)$' % '|'.join(re.escape(d) for d in allowed_domains if d is not None) + message = ("allowed_domains accepts only domains, not URLs. " + "Ignoring URL entry %s in allowed_domains." % domain) + warnings.warn(message, URLWarning) + domains = [re.escape(d) for d in allowed_domains if d is not None] + regex = r'^(.*\.)?(%s)$' % '|'.join(domains) return re.compile(regex) def spider_opened(self, spider): diff --git a/tests/test_contracts.py b/tests/test_contracts.py index b4209e1f6..3571d8f7f 100644 --- a/tests/test_contracts.py +++ b/tests/test_contracts.py @@ -5,12 +5,13 @@ from twisted.internet import defer from twisted.python import failure from twisted.trial import unittest +from scrapy import FormRequest from scrapy.crawler import CrawlerRunner from scrapy.spidermiddlewares.httperror import HttpError from scrapy.spiders import Spider from scrapy.http import Request from scrapy.item import Item, Field -from scrapy.contracts import ContractsManager +from scrapy.contracts import ContractsManager, Contract from scrapy.contracts.default import ( UrlContract, ReturnsContract, @@ -28,6 +29,15 @@ class ResponseMock(object): url = 'http://scrapy.org' +class CustomFormContract(Contract): + name = 'custom_form' + request_cls = FormRequest + + def adjust_request_args(self, args): + args['formdata'] = {'name': 'scrapy'} + return args + + class TestSpider(Spider): name = 'demo_spider' @@ -104,13 +114,20 @@ class TestSpider(Spider): """ pass + def custom_form(self, response): + """ + @url http://scrapy.org + @custom_form + """ + pass + class InheritsTestSpider(TestSpider): name = 'inherits_demo_spider' class ContractsManagerTest(unittest.TestCase): - contracts = [UrlContract, ReturnsContract, ScrapesContract] + contracts = [UrlContract, ReturnsContract, ScrapesContract, CustomFormContract] def setUp(self): self.conman = ContractsManager(self.contracts) @@ -241,6 +258,12 @@ class ContractsManagerTest(unittest.TestCase): self.assertEqual(crawler.spider.visited, 2) + def test_form_contract(self): + spider = TestSpider() + request = self.conman.from_method(spider.custom_form, self.results) + self.assertEqual(request.method, 'POST') + self.assertIsInstance(request, FormRequest) + def test_inherited_contracts(self): spider = InheritsTestSpider() diff --git a/tests/test_engine.py b/tests/test_engine.py index 719c0c60c..856465161 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -103,6 +103,7 @@ class CrawlerRun(object): self.respplug = [] self.reqplug = [] self.reqdropped = [] + self.reqreached = [] self.itemerror = [] self.itemresp = [] self.signals_catched = {} @@ -124,6 +125,7 @@ class CrawlerRun(object): self.crawler.signals.connect(self.item_error, signals.item_error) self.crawler.signals.connect(self.request_scheduled, signals.request_scheduled) self.crawler.signals.connect(self.request_dropped, signals.request_dropped) + self.crawler.signals.connect(self.request_reached, signals.request_reached_downloader) self.crawler.signals.connect(self.response_downloaded, signals.response_downloaded) self.crawler.crawl(start_urls=start_urls) self.spider = self.crawler.spider @@ -155,6 +157,9 @@ class CrawlerRun(object): def request_scheduled(self, request, spider): self.reqplug.append((request, spider)) + def request_reached(self, request, spider): + self.reqreached.append((request, spider)) + def request_dropped(self, request, spider): self.reqdropped.append((request, spider)) @@ -212,6 +217,8 @@ class EngineTest(unittest.TestCase): responses_count = len(self.run.respplug) self.assertEqual(scheduled_requests_count, dropped_requests_count + responses_count) + self.assertEqual(len(self.run.reqreached), + responses_count) def _assert_dropped_requests(self): self.assertEqual(len(self.run.reqdropped), 1) @@ -219,6 +226,7 @@ class EngineTest(unittest.TestCase): def _assert_downloaded_responses(self): # response tests self.assertEqual(8, len(self.run.respplug)) + self.assertEqual(8, len(self.run.reqreached)) for response, _ in self.run.respplug: if self.run.getpath(response.url) == '/item999.html':