From 91eea82eefbc60e8c9824faf49fb510ad4aabc50 Mon Sep 17 00:00:00 2001 From: Pablo Hoffman Date: Sat, 8 Aug 2009 07:26:20 -0300 Subject: [PATCH] added XPathLoader for working with XPath Selectors more conveniently --- docs/experimental/loaders.rst | 135 ++++++++++++++++++++---------- docs/experimental/newitems.rst | 2 +- docs/topics/items.rst | 2 +- scrapy/newitem/loader/__init__.py | 26 +++++- scrapy/tests/test_itemloader.py | 26 +++++- 5 files changed, 144 insertions(+), 47 deletions(-) diff --git a/docs/experimental/loaders.rst b/docs/experimental/loaders.rst index a4350d89c..f6632832d 100644 --- a/docs/experimental/loaders.rst +++ b/docs/experimental/loaders.rst @@ -37,55 +37,58 @@ Reducer. Here is a typical Loader usage in a :ref:`Spider ` using the :ref:`Product item defined in the Items chapter `.:: - from scrapy.item.loader import Loader + from scrapy.item.loader import XPathLoader from scrapy.xpath import HtmlXPathSelector from myproject.items import Product def parse(self, response): - x = HtmlXPathSelector(response) - l = Loader(item=Product()) - l.add_value('name', x.x('//div[@class="product_name"]').extract()) - l.add_value('name', x.x('//div[@class="product_title"]').extract()) - l.add_value('price', x.x('//p[@id="price"]').extract()) - l.add_value('stock', x.x('//p[@id="stock"]').extract()) - l.add_value('last_updated', 'today') + l = XPathLoader(item=Product(), response=response) + l.add_xpath('name', '//div[@class="product_name"]') + l.add_xpath('name', '//div[@class="product_title"]') + l.add_xpath('price', '//p[@id="price"]') + l.add_xpath('stock', '//p[@id="stock"]') + l.add_value('last_updated', 'today') # you can also literal values return l.get_item() -By looking at that code we can see the ``name`` field is being extracted from -two different XPath locations in the page: +By quickly looking at that code we can see the ``name`` field is being +extracted from two different XPath locations in the page: * ``//div[@class="product_name"]`` * ``//div[@class="product_title"]`` -So both XPaths are used for extracting data from the page, and the data -returned by them is collected to be assigned to the ``name`` attribute. +In other words, data is being collected by extracting it from two XPath +locations, using the :meth:`~XPathLoader.add_xpath` method. This is the data +that will be assigned to the ``name`` field later. Afterwards, similar calls are used for ``price`` and ``stock`` fields, and finally the ``last_update`` field is populated directly with a literal value -(``today``). +(``today``) using a different method: :meth:`~Loader.add_value`. Finally, when all data is collected, the :meth:`Loader.get_item` method is called which actually populates and returns the item populated with the data -previously extracted with the ``add_value`` calls. +previously extracted and collected with the :meth:`~XPathLoader.add_xpath` and +:meth:`~Loader.add_value` calls. + +.. _topics-loader-expred: Expanders and Reducers ====================== -A Loader is composed of one expander and reducer for each item field. The -Expander processes the extracted data as soon as it's received through the -:meth:`Loader.add_value` method, and the result of the expander is collected -and kept inside the Loader. After collecting all data, the -:meth:`Loader.get_item` method is called to actually populate and get the Item. -That's when the Reducers are called with the data previously collected (using -the Expanders) and the output of the Reducers are the actual values that get -assigned to the item. +A Loader is composed of one expander and one reducer for each item field. The +Expander processes the extracted data as soon as it's received (through the +:meth:`~XPathLoader.add_xpath` or :meth:`~Loader.add_value` methods) and the +result of the expander is collected and kept inside the Loader. After +collecting all data, the :meth:`Loader.get_item` method is called to actually +populate and get the Item. That's when the Reducers are called with the data +previously collected (using the Expanders) and the output of the Reducers are +the actual values that get assigned to the item. Let's see an example to illustrate how Expanders and Reducers are called, for a particular field (the same applies for any other field):: - l = Loader(Product()) - l.add_value('name', x.x(xpath1).extract()) # (1) - l.add_value('name', x.x(xpath2).extract()) # (2) + l = XPathLoader(Product(), some_selector) + l.add_xpath('name', xpath1) # (1) + l.add_xpath('name', xpath2) # (2) return l.get_item() # (3) So what happens is: @@ -150,14 +153,14 @@ value and extracts a length from it:: return parsed_length Since it receives a ``loader_args`` the Expander will pass the currently active -loader arguments when calling it. +Loader arguments when calling it. -There are seveal ways to pass loader arguments: +There are seveal ways to pass Loader arguments: 1. Passing arguments on Loader declaration:: class ProductLoader(Loader): - length_exp = TreeExpander(parse_length, unit='cm') + length_exp = TreeExpander(parse_length, unit='cm') 2. Passing arguments on Loader instantiation:: @@ -165,7 +168,7 @@ There are seveal ways to pass loader arguments: 3. Passing arguments on Loader usage:: - l.add_value('length', x.x('//div').extract(), unit='cm') + l.add_xpath('length', '//div', unit='cm') Loader objects ============== @@ -176,16 +179,23 @@ Loader objects given, one is instantiated using the class in :attr:`default_item_class`. .. method:: add_value(field_name, value, \**new_loader_args) - - Add the given ``value`` for the given field. - - The value is passed through the field expander and its output appened - to the data collected for that field. If the field already contains - collected data, the new data is added. + + Add the given ``value`` for the given field. + + The value is passed through the :ref:`field expander + ` and its output appened to the data collected + for that field. If the field already contains collected data, the new + data is added. If any keyword arguments are passed, they're used as :ref:`Loader arguments ` when calling the expanders. + Examples:: + + loader.add_value('name', u'Color TV') + loader.add_value('colours', [u'white', u'blue']) + loader.add_value('length', u'100', default_unit='cm') + .. method:: replace_value(field_name, value, \**new_loader_args) Similar to :meth:`add_value` but replaces collected data instead of @@ -194,7 +204,10 @@ Loader objects .. method:: get_item() - Populate the item with the data collected so far, and return it. + Populate the item with the data collected so far, and return it. The + data collected is first passed through the :ref:`field reducers + ` to get the final value to assign to each item + field. .. method:: get_expanded_value(field_name) @@ -229,6 +242,41 @@ Loader objects The default reducer to use for those fields which don't define a specific expander +.. class:: XPathLoader([item, selector, response], \**loader_args) + + The :class:`XPathLoader` class extends the :class:`Loader` class providing + more convenient mechanisms for extracting data from web pages using + :ref:`XPath selectors `. + + :class:`XPathLoader` objects accept two more additional parameters in their + constructors: + + :param selector: The selector to extract data from, when using the + :meth:`add_xpath` or :meth:`replace_xpath` method. + :type selector: :class:`~scrapy.xpath.XPathSelector` object + + :param response: The response used to construct the selector using the + :attr:`default_selector_class`, unless the selector argument is given, + in which case this argument is ignored. + :type response: :class:`~scrapy.http.Response` object + + .. attribute:: default_selector_class + + The class used to construct the selector, if only a response is given + in the constructor + + .. method:: add_xpath(field_name, xpath, \**new_loader_args) + + Similar to :meth:`Loader.add_value` but receives an XPath instead of a + value, which is used to extract a list of unicode strings from the + selector associated with this :class:`XPathLoader`. + + .. method:: replace_xpath(field_name, xpath, \**new_loader_args) + + Similar to :meth:`add_xpath` but replaces collected data instead of + adding it. + + Reusing and extending Loaders ============================= @@ -283,7 +331,7 @@ one, as the other is just an identity expander. .. module:: scrapy.newitem.loader.expanders :synopsis: Expander classes to use with Item Loaders - + .. class:: TreeExpander(\*functions, \**default_loader_arguments) An expander which applies the given functions consecutively, in order, to @@ -297,11 +345,12 @@ one, as the other is just an identity expander. is why this expander is called Tree Expander. Each expander function can optionally receive a ``loader_args`` argument, - which will contain the currently active loader arguments. + which will contain the currently active :ref:`Loader arguments + `. The keyword arguments passed in the consturctor are used as the default - loader arguments passed to on each expander call. This arguments can be - overriden with specific loader arguments passed on each expander call. + Loader arguments passed to on each expander call. This arguments can be + overriden with specific noader arguments passed on each expander call. IdentityExpander ---------------- @@ -337,7 +386,7 @@ or callable as reducer. Return the values to reduce unchanged, so it's used for multi-valued fields. It doesn't receive any constructor arguments. - + Example:: features_red = Identity() @@ -345,13 +394,13 @@ or callable as reducer. .. class:: Join(separator=u' ') Return a the values to reduce joined with the separator given in the - constructor, which defaults to ``u' '``. + constructor, which defaults to ``u' '``. When using the default separator, this reducer is equivalent to the function: ``u' '.join`` Examples:: - + name_red = Join() name_red = Join('
') diff --git a/docs/experimental/newitems.rst b/docs/experimental/newitems.rst index 0cd3c88ba..8ee880018 100644 --- a/docs/experimental/newitems.rst +++ b/docs/experimental/newitems.rst @@ -65,7 +65,7 @@ Working with Items ================== Here are some examples of common tasks performed with items, using the -``Product`` item :ref:`declared above `. You will +``Product`` item :ref:`declared above `. You will notice the API is very similar to the `dict API`_. Creating items diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 4fbfe2870..c607cf76e 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -57,7 +57,7 @@ RobustScrapedItems .. warning:: RobustScapedItems are deprecated and will be replaced by the :ref:`New item - API ` (still in development). + API ` (still in development). .. module:: scrapy.contrib.item :synopsis: Objects for storing scraped data diff --git a/scrapy/newitem/loader/__init__.py b/scrapy/newitem/loader/__init__.py index cf8fe36a0..9edebf061 100644 --- a/scrapy/newitem/loader/__init__.py +++ b/scrapy/newitem/loader/__init__.py @@ -4,6 +4,7 @@ from scrapy.utils.datatypes import MergeDict from scrapy.newitem import Item from scrapy.newitem.loader.reducers import TakeFirst from scrapy.newitem.loader.expanders import IdentityExpander +from scrapy.xpath import HtmlXPathSelector class Loader(object): @@ -12,7 +13,7 @@ class Loader(object): default_reducer = TakeFirst() def __init__(self, item=None, **loader_args): - if not item: + if item is None: item = self.default_item_class() self._item = loader_args['item'] = item self._loader_args = loader_args @@ -59,3 +60,26 @@ class Loader(object): loader_args = MergeDict(new_loader_args, self._loader_args) expander = self.get_expander(field_name) return expander(value, loader_args=loader_args) + + +class XPathLoader(Loader): + + default_selector_class = HtmlXPathSelector + + def __init__(self, item=None, selector=None, response=None, **loader_args): + if selector is None and response is None: + raise RuntimeError("%s must be instantiated with a selector" \ + "or response" % self.__class__.__name__) + if selector is None: + selector = self.default_selector_class(response) + self._selector = selector + loader_args.update(selector=selector, response=response) + super(XPathLoader, self).__init__(item, **loader_args) + + def add_xpath(self, field_name, xpath, **new_loader_args): + self.add_value(field_name, self._selector.x(xpath).extract(), \ + **new_loader_args) + + def replace_xpath(self, field_name, xpath, **new_loader_args): + self.replace_value(field_name, self._selector.x(xpath).extract(), \ + **new_loader_args) diff --git a/scrapy/tests/test_itemloader.py b/scrapy/tests/test_itemloader.py index b674e3f50..fdfb32458 100644 --- a/scrapy/tests/test_itemloader.py +++ b/scrapy/tests/test_itemloader.py @@ -1,9 +1,11 @@ import unittest -from scrapy.newitem.loader import Loader +from scrapy.newitem.loader import Loader, XPathLoader from scrapy.newitem.loader.expanders import TreeExpander, IdentityExpander from scrapy.newitem.loader.reducers import Join, Identity from scrapy.newitem import Item, Field +from scrapy.xpath import HtmlXPathSelector +from scrapy.http import HtmlResponse # test items @@ -226,5 +228,27 @@ class LoaderTest(unittest.TestCase): il = TestLoader() self.assertRaises(KeyError, il.add_value, 'wrong_field', [u'lala', u'lolo']) + +class TestXPathLoader(XPathLoader): + default_item_class = TestItem + name_exp = TreeExpander(lambda v: v.title()) + +class XPathLoaderTest(unittest.TestCase): + + def test_constructor_errors(self): + self.assertRaises(RuntimeError, XPathLoader) + + def test_constructor_with_selector(self): + sel = HtmlXPathSelector(text=u"
marta
") + l = TestXPathLoader(selector=sel) + l.add_xpath('name', '//div/text()') + self.assertEqual(l.get_reduced_value('name'), u'Marta') + + def test_constructor_with_response(self): + response = HtmlResponse(url="", body="
marta
") + l = TestXPathLoader(response=response) + l.add_xpath('name', '//div/text()') + self.assertEqual(l.get_reduced_value('name'), u'Marta') + if __name__ == "__main__": unittest.main()