From 04fc0471ec89e47dda802b2336d9deb7092c0127 Mon Sep 17 00:00:00 2001 From: Ismael Carnales Date: Thu, 29 Jan 2009 12:29:28 +0000 Subject: [PATCH] first version of the new tutorial (work in progress) --HG-- extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40792 --- scrapy/trunk/docs/tutorial.rst | 310 +++++++++++++++++++++++++++++++++ 1 file changed, 310 insertions(+) create mode 100644 scrapy/trunk/docs/tutorial.rst diff --git a/scrapy/trunk/docs/tutorial.rst b/scrapy/trunk/docs/tutorial.rst new file mode 100644 index 000000000..09524ba46 --- /dev/null +++ b/scrapy/trunk/docs/tutorial.rst @@ -0,0 +1,310 @@ +.. _tutorial: + +=============== +Scrapy Tutorial +=============== + +In this tutorial, we'll assume that Scrapy is already installed in your system, +if not see :ref:`intro-install`. + +We are going to use `Open directory project (dmoz) `_ as +our example domain to scrape. + +Creating a project +================== + +Before start scraping, you will have set up a new Scrapy project. Enter a +directory where you'd like to store your code and then run:: + + scrapy-admin.py startproject dmoz + +This will create a ``dmoz`` directory with the following contents:: + + dmoz/ + manage.py + dmoz/ + __init__.py + items.py + settings.py + spiders/ + __init__.py + templates/ + ... + +These are basically: + +* ``manage.py``: the project's control script. +* ``dmoz/``: the project's python module, you'll later import your code from + here. +* ``dmoz/settings.py``: the project's settings file. +* ``dmoz/spiders/``: a directory where you'll later put your spiders. + +We'll talk more about the project files later, now let's go into spiders. + +Spiders +======= + +Spiders are custom modules written by you, the user, to scrape information from +a certain domain. Their duty is to feed the Scrapy engine with URLs to +download, and then parse the downloaded contents in the search for data or more +URLs to follow. + +They are the heart of a Scrapy project and where most part of the action takes +place. + +To create our first spider, save this code in a file named dmoz.py inside +*dmoz/spiders* folder:: + + from scrapy.spider import BaseSpider + + class OpenDirectorySpider(BaseSpider): + domain_name = "dmoz.org" + start_urls = [ + "http://www.dmoz.org/Computers/Programming/Languages/Python/Books/", + "http://www.dmoz.org/Computers/Programming/Languages/Python/Resources/" + ] + + def parse(self, response): + filename = response.url.split("/")[-2] + open(filename, 'w').write(response.body) + return [] + + SPIDER = OpenDirectorySpider() + +The first line imports the class BaseSpider. For the purpose of creating a +working spider, you must subclass BaseSpider, and then define the three main, +mandatory, attributes: + +* ``domain_name``: identifies the spider. It must be unique, that is, you can't + set the same domain name for different spiders. + +* ``start_urls``: is a list + of URLs where the spider will begin to crawl from. + So, the first pages downloaded will be those listed here. The subsequent URLs + will be generated successively from data contained in the start URLs. + +* ``parse`` is the callback method of the spider. This means that each time a + URL is retrieved, the downloaded data (response) will be passed to this method. + + The ``parse`` method is in charge of processing the response and returning + scraped data and or more URLs to follow, because of this, the method must + always return a list or at least an empty one. + +In the last line, we instantiate our spider class. + +Crawling +======== + +To put our spider to work, go to the project's top level directory and run:: + + ./scrapy-ctl.py crawl dmoz.org + +The ``crawl dmoz.org`` subcommand runs the spider for the ``dmoz.org`` domain, you'll get an output like this:: + + [-] Log opened. + [dmoz] INFO: Enabled extensions: TelnetConsole, WebConsole + [dmoz] INFO: Enabled downloader middlewares: ErrorPagesMiddleware, CookiesMiddleware, HttpAuthMiddleware, UserAgentMiddleware, RetryMiddleware, CommonMiddleware, RedirectMiddleware, CompressionMiddleware + [dmoz] INFO: Enabled spider middlewares: OffsiteMiddleware, RefererMiddleware, UrlLengthMiddleware, DepthMiddleware, UrlFilterMiddleware + [dmoz] INFO: Enabled item pipelines: + [-] scrapy.management.web.WebConsole starting on 60738 + [-] scrapy.management.telnet.TelnetConsole starting on 51506 + [dmoz/dmoz.org] INFO: Domain opened + [dmoz/dmoz.org] DEBUG: Crawled from + [dmoz/dmoz.org] DEBUG: Crawled from + [dmoz/dmoz.org] INFO: Domain closed (finished) + [scrapy.management.web.WebConsole] (Port 60738 Closed) + [scrapy.management.telnet.TelnetConsole] (Port 51506 Closed) + [-] Main loop terminated. + +Pay attention to the lines labeled ``[dmoz/dmoz.org]``, which corresponds to +our spider identified by the domain "dmoz.org". You can see a log line for each +URL defined in ``start_urls``. Because these URLs are the starting ones, they +have no referrers, and this condition is indicated at the end of the log line, +where it says ``from ``. + +But more interesting, as our ``parse`` method instructs, two files have been +created: *Books* and *Resources*, with the content of both URLs. + +Shell +===== + +Scrapy comes with an built-in shell that ... XXX ... + +To use this feature you must have IPython installed on your system. + +IPython is an extended python console, and the ``shell`` command sets the +Python path, imports some important Scrapy libraries and sets some useful local +variables for you to play with. + +To start a shell you must go to the project's top level directory and run:: + + ./scrapy-ctl.py shell http://www.dmoz.org/Computers/Programming/Languages/Python/Books/ + +This is what the shell looks like:: + + [-] Log opened. + Scrapy 0.7.0 - Interactive scraping console + + [-] scrapy.management.web.WebConsole starting on 33227 + [-] scrapy.management.telnet.TelnetConsole starting on 42311 + Downloading URL... Done. + ------------------------------------------------------------------------------ + Available local variables: + xxs: + url: http://www.dmoz.org/Computers/Programming/Languages/Python/Books/ + spider: + hxs: + item: + response: + Available commands: + get : Fetches an url and updates all variables. + scrapehelp: Prints this help. + ------------------------------------------------------------------------------ + Python 2.6.1 (r261:67515, Dec 7 2008, 08:27:41) + Type "copyright", "credits" or "license" for more information. + + IPython 0.9.1 -- An enhanced Interactive Python. + ? -> Introduction and overview of IPython's features. + %quickref -> Quick reference. + help -> Python's own help system. + object? -> Details about 'object'. ?object also works, ?? prints more. + + In [1]: + +After the shell loads, it will put the result of the request action for the +given URL in a ``response`` variable, so if you enter ``response.body`` the +downloaded data will be printed on the screen. + +The shell has also instantiated for two selectors with this respose as an +initialization parameter, let's see what selectors are for. + +Selectors +========= + +In order to extract information from web pages Scrapy adopted `XPath +`_, a language for finding information in a XML +document navigating trough its elements and attributes. + +Here are some examples of XPath queries and their corresponding results: + +* ``/html/head/title``: Will give you the ``title`` node of the document. +* ``/html/head/title/text()``: Will give you the text inside the ``title`` node of the document. +* ``//td``: Will select all the ``td`` elements. +* ``//div[@class="queryMe"]``: Will select all the ``div`` elements with ``class = queryMe``. + +This are really simple examples of what you can do with XPath, we strongly +suggest you to follow this `XPath tutorial +`_ before continuing. + +----- + +Scrapy defines a XPathSelector class that comes in two flavours, +HtmlXPatSelector (for HTML) and XmlXPathSelector (for XML), in order to use +them you must instantiate the desired class with a Response object. + +When you've opened a shell (if not, go back and open one, we're going to use +it), it has automatically arranged two selectors for you: ``xxs`` and ``hxs``, +``xxs`` is an XML selector and ``hxs`` is an HTML one, we'll use the ``hxs`` +selector in this example. + +You can see selectors as objects that represents nodes in the document +structure. So, these instantiated selectors are associated to the root node, or +the entire document. + +Selectors have three methods: ``x``, ``extract`` and ``re``. + +* ``x``: returns a list of selectors, each of them representing the nodes + gotten in the xpath expression given as parameter. +* ``extract``: actually extracts the data contained in the node. Does not + receive parameters. +* ``re``: returns a list of results of a regular expression given as parameter. + +So let's try them in our console:: + + In [1]: hxs.x('/html/head/title') + Out[1]: [] + + In [2]: hxs.x('/html/head/title').extract() + Out[2]: [u'Open Directory - Computers: Programming: Languages: Python: Books'] + + In [3]: hxs.x('/html/head/title/text()') + Out[3]: [] + + In [4]: hxs.x('/html/head/title/text()').extract() + Out[4]: [u'Open Directory - Computers: Programming: Languages: Python: Books'] + + In [5]: hxs.x('/html/head/title/text()').re('(\w+):') + Out[5]: [u'Computers', u'Programming', u'Languages', u'Python'] + +Now, let's try to extract the sites information from the directory page. + +If you do a ``response.body`` in the console, look at the source code of the +page or better yet use Firebug to inspect the page, you'll find that the sites +part of the code is an ``ul`` tag, in fact the *second* ``ul`` tag. + +So we can select each ``li`` item belonging to the sites list with this code:: + + hxs.x('//ul[2]/li') + +And from them, the sites descriptions:: + + hxs.x('//ul[2]/li/text()').extract() + +The sites titles:: + + hxs.x('//ul[2]/li/a/text()').extract() + +And the sites links:: + + hxs.x('//ul[2]/li/a/@href').extract() + +As we said before, each ``x()`` call returns a list of selectors, so we can +concatenate further ``x()`` calls to dig deeper into a node. We are goin to use +that property here, so:: + + sites = hxs.x('//ul[2]/li') + for site in sites: + title = site.x('a/text()').extract() + link = site.x('a/@href').extract() + desc = site.x('text()').extract() + print title, link, desc + +Let's add this code to our spider:: + + from scrapy.spider import BaseSpider + from scrapy.xpath.selector import HtmlXPathSelector + + + class OpenDirectorySpider(BaseSpider): + domain_name = "dmoz.org" + start_urls = [ + "http://www.dmoz.org/Computers/Programming/Languages/Python/Books/", + "http://www.dmoz.org/Computers/Programming/Languages/Python/Resources/" + ] + + def parse(self, response): + hxs = HtmlXPathSelector(response) + sites = hxs.x('//ul[2]/li') + for site in sites: + title = site.x('a/text()').extract() + link = site.x('a/@href').extract() + desc = site.x('text()').extract() + print title, link, desc + return [] + + SPIDER = OpenDirectorySpider() + +Now try crawling the dmoz.org domain again and you'll see sites being printed +in your output, run:: + + ./scrapy-ctl.py crawl dmoz.org + +Items +===== + +In Scrapy, items are the placeholder to use for the scraped data. They are +represented by a ScrapedItem object, or any descendant class instance, and +store the information in class attributes + +Item Pipeline +=============