From 2e01fa166759266fe050fa2aa5952a86e3e08e59 Mon Sep 17 00:00:00 2001 From: elpolilla Date: Mon, 15 Dec 2008 13:44:42 +0000 Subject: [PATCH] Added part of the new scrapy tutorial --HG-- extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40498 --- sites/scrapy.org/docs/sources/index.rst | 1 + .../docs/sources/tutorial2/index.rst | 8 ++++ .../docs/sources/tutorial2/tutorial1.rst | 29 ++++++++++++++ .../docs/sources/tutorial2/tutorial2.rst | 38 +++++++++++++++++++ 4 files changed, 76 insertions(+) create mode 100644 sites/scrapy.org/docs/sources/tutorial2/index.rst create mode 100644 sites/scrapy.org/docs/sources/tutorial2/tutorial1.rst create mode 100644 sites/scrapy.org/docs/sources/tutorial2/tutorial2.rst diff --git a/sites/scrapy.org/docs/sources/index.rst b/sites/scrapy.org/docs/sources/index.rst index 0172bc036..59a7898b1 100644 --- a/sites/scrapy.org/docs/sources/index.rst +++ b/sites/scrapy.org/docs/sources/index.rst @@ -12,3 +12,4 @@ Contents: basics scrapy_intro tutorial/index + tutorial2/index diff --git a/sites/scrapy.org/docs/sources/tutorial2/index.rst b/sites/scrapy.org/docs/sources/tutorial2/index.rst new file mode 100644 index 000000000..944f5600c --- /dev/null +++ b/sites/scrapy.org/docs/sources/tutorial2/index.rst @@ -0,0 +1,8 @@ +======== +Tutorial +======== + +.. toctree:: + + tutorial1 + tutorial2 diff --git a/sites/scrapy.org/docs/sources/tutorial2/tutorial1.rst b/sites/scrapy.org/docs/sources/tutorial2/tutorial1.rst new file mode 100644 index 000000000..2aadbd53a --- /dev/null +++ b/sites/scrapy.org/docs/sources/tutorial2/tutorial1.rst @@ -0,0 +1,29 @@ +===================== +Setting everything up +===================== + +| In this tutorial, we'll teach you how to scrape http://www.dmoz.org, a websites directory. +| We'll assume that you've checked and fulfilled the requirements specified in the Download page, and that Scrapy is already installed in your system. + +For starting a new project, enter the directory where you'd like your project to be located, and run:: + + scrapy-admin.py startproject dmoz + +As long as Scrapy is well installed and the path is set, this should create a directory called "dmoz" +containing the following files: + +* *scrapy-ctl.py* - the project's control script. It's used for running the different tasks (like "crawl" and "parse"). We'll talk more about this later. +* *scrapy_settings.py* - the project's settings file. +* *items.py* - were you define the different kinds of items you're going to scrape. +* *spiders* - directory where you'll later place your spiders. +* *templates* - directory containing some templates for newly created spiders, and where you can put your own. + +| Ok, now that you have your project's structure defined, the last thing to do is to set your PYTHONPATH to your project's directory. +| You can do this by adding this to your .bashrc file: + +:: + + export PYTHONPATH=/path/to/your/project + + +Now you can continue with the next part of the tutorial. diff --git a/sites/scrapy.org/docs/sources/tutorial2/tutorial2.rst b/sites/scrapy.org/docs/sources/tutorial2/tutorial2.rst new file mode 100644 index 000000000..9539d4b77 --- /dev/null +++ b/sites/scrapy.org/docs/sources/tutorial2/tutorial2.rst @@ -0,0 +1,38 @@ +================ +Our first spider +================ + +| Ok, the time to write our first spider has come. +| Make sure you're standing on your project's directory and run: + +:: + + ./scrapy-ctl genspider dmoz dmoz.org + +This should create a file called dmoz.py under the *spiders* directory looking similar to this:: + + # -*- coding: utf8 -*-< + import re + from scrapy.xpath import HtmlXPathSelector + from scrapy.item import ScrapedItem + from scrapy.link.extractors import RegexLinkExtractor + from scrapy.contrib.spiders import CrawlSpider, Rule + + class DmozSpider(CrawlSpider): + domain_name = "dmoz.org" + start_urls = ['http://www.dmoz.org/'] + + rules = ( + Rule(RegexLinkExtractor(allow=(r'Items/', ), 'parse_item', follow=True) + ) + + def parse_item(self, response): + #xs = HtmlXPathSelector(response) + #i = ScrapedItem() + #i.attribute('site_id', xs.x("//input[@id="sid"]/@value")) + #i.attribute('name', xs.x("//div[@id='name']")) + #i.attribute('description', xs.x("//div[@id='description']")) + #return [i] + + SPIDER = DmozSpider() +