diff --git a/scrapy/trunk/scrapy/tests/test_xpath.py b/scrapy/trunk/scrapy/tests/test_xpath.py
index 3a8e6dd78..57600b786 100644
--- a/scrapy/trunk/scrapy/tests/test_xpath.py
+++ b/scrapy/trunk/scrapy/tests/test_xpath.py
@@ -191,5 +191,35 @@ class XPathTestCase(unittest.TestCase):
self.assertEqual(xxs.extract(),
u'lala')
-if __name__ == "__main__":
- unittest.main()
+ def test_unquote(self):
+ xmldoc='\n'.join((
+ '',
+ ' lala',
+ ' ',
+ ' blabla&moreatestoh',
+ ' PPPPppp&la]]>',
+ ' ',
+ ' pff',
+ ''))
+ xxs = XmlXPathSelector(text=xmldoc)
+
+ self.assertEqual(xxs.extract_unquoted(), u'')
+
+ self.assertEqual(xxs.x('/root').extract_unquoted(), [u''])
+ self.assertEqual(xxs.x('/root/text()').extract_unquoted(), [
+ u'\n lala\n ',
+ u'\n pff\n'])
+
+ self.assertEqual(xxs.x('//*').extract_unquoted(), [u'', u'', u''])
+ self.assertEqual(xxs.x('//text()').extract_unquoted(), [
+ u'\n lala\n ',
+ u'\n blabla&more',
+ u'a',
+ u'test',
+ u'oh\n ',
+ u'lalalal&pppppPPPPppp&la',
+ u'\n ',
+ u'\n pff\n'])
+
+ if __name__ == "__main__":
+ unittest.main()
diff --git a/scrapy/trunk/scrapy/xpath/selector.py b/scrapy/trunk/scrapy/xpath/selector.py
index f49e5c84c..3902efe27 100644
--- a/scrapy/trunk/scrapy/xpath/selector.py
+++ b/scrapy/trunk/scrapy/xpath/selector.py
@@ -73,6 +73,13 @@ class XPathSelector(object):
text = unicode(self.xmlNode)
return text
+ def extract_unquoted(self):
+ """Get unescaped contents from the text node (no entities, no CDATA)"""
+ if self.x('self::text()'):
+ return unicode(self.xmlNode.getContent(), errors='ignore')
+ else:
+ return u''
+
def register_namespace(self, prefix, uri):
"""Register namespace so that it can be used in XPath queries"""
self.doc.xpathContext.xpathRegisterNs(prefix, uri)
@@ -98,12 +105,15 @@ class XPathSelectorList(list):
"""Perform the re() method on each XPathSelector of the list, and
return the result as a flattened list of unicode strings"""
return flatten([x.re(regex) for x in self])
-
+
def extract(self):
"""Return a list of unicode strings with the content referenced by each
XPathSelector of the list"""
return [x.extract() if isinstance(x, XPathSelector) else x for x in self]
+ def extract_unquoted(self):
+ return [x.extract_unquoted() if isinstance(x, XPathSelector) else x for x in self]
+
class XmlXPathSelector(XPathSelector):
"""XPathSelector for XML content"""