Cache response's element trees using LxmlDocument similar to Libxml2Document

This commit is contained in:
Daniel Graña 2012-04-13 14:43:30 -03:00
parent ac4f6cc17c
commit 4c7d29b7f7
3 changed files with 84 additions and 55 deletions

View File

@ -0,0 +1,31 @@
"""
This module contains a simple class (LxmlDocument) which provides cache and
garbage collection to lxml element tree documents.
"""
import weakref
from lxml import etree
from scrapy.utils.trackref import object_ref
def _factory(response, parser_cls):
url = response.url
body = response.body_as_unicode().strip().encode('utf8') or '<html/>'
parser = parser_cls(recover=True, encoding='utf8')
return etree.fromstring(body, parser=parser, base_url=url)
class LxmlDocument(object_ref):
cache = weakref.WeakKeyDictionary()
__slots__ = ['xmlDoc', 'xpathContext', '__weakref__']
def __new__(cls, response, parser=etree.HTMLParser):
cache = cls.cache.setdefault(response, {})
if parser not in cache:
obj = object_ref.__new__(cls)
cache[parser] = _factory(response, parser)
return cache[parser]
def __str__(self):
return "<LxmlDocument %s>" % self.root.tag

View File

@ -9,42 +9,35 @@ from scrapy.utils.trackref import object_ref
from scrapy.utils.python import unicode_to_str
from scrapy.utils.decorator import deprecated
from scrapy.http import TextResponse
from .lxmldocument import LxmlDocument
from .list import XPathSelectorList
__all__ = ['HtmlXPathSelector', 'XmlXPathSelector', 'XPathSelector', \
'XPathSelectorList']
'XPathSelectorList']
class XPathSelector(object_ref):
__slots__ = ['response', 'text', 'expr', 'namespaces', '_root', '_xpathev', \
'__weakref__']
__slots__ = ['response', 'text', 'namespaces', '_expr', '_root', '__weakref__']
_parser = etree.HTMLParser
_tostring_method = 'html'
def __init__(self, response=None, text=None, root=None, expr=None, namespaces=None):
if text:
self.response = TextResponse(url='about:blank', \
def __init__(self, response=None, text=None, namespaces=None, _root=None, _expr=None):
if text is not None:
response = TextResponse(url='about:blank', \
body=unicode_to_str(text, 'utf-8'), encoding='utf-8')
else:
self.response = response
self._root = root
self._xpathev = None
self.namespaces = namespaces
self.expr = expr
if response is not None:
_root = LxmlDocument(response, self._parser)
@property
def root(self):
if self._root is None:
url = self.response.url
body = self.response.body_as_unicode().strip().encode('utf8') or '<html/>'
parser = self._parser(recover=True, encoding='utf8')
self._root = etree.fromstring(body, parser=parser, base_url=url)
assert self._root is not None, 'BUG lxml selector with None root'
return self._root
self.namespaces = namespaces
self.response = response
self._root = _root
self._expr = _expr
def select(self, xpath):
try:
xpathev = self.root.xpath
xpathev = self._root.xpath
except AttributeError:
return XPathSelectorList([])
@ -56,7 +49,7 @@ class XPathSelector(object_ref):
if type(result) is not list:
result = [result]
result = [self.__class__(root=x, expr=xpath, namespaces=self.namespaces)
result = [self.__class__(_root=x, _expr=xpath, namespaces=self.namespaces)
for x in result]
return XPathSelectorList(result)
@ -65,15 +58,15 @@ class XPathSelector(object_ref):
def extract(self):
try:
return etree.tostring(self.root, method=self._tostring_method, \
return etree.tostring(self._root, method=self._tostring_method, \
encoding=unicode)
except (AttributeError, TypeError):
if self.root is True:
if self._root is True:
return u'1'
elif self.root is False:
elif self._root is False:
return u'0'
else:
return unicode(self.root)
return unicode(self._root)
def register_namespace(self, prefix, uri):
if self.namespaces is None:
@ -85,7 +78,7 @@ class XPathSelector(object_ref):
def __str__(self):
data = repr(self.extract()[:40])
return "<%s xpath=%r data=%s>" % (type(self).__name__, self.expr, data)
return "<%s xpath=%r data=%s>" % (type(self).__name__, self._expr, data)
__repr__ = __str__

View File

@ -2,35 +2,40 @@
Selectors tests, specific for lxml backend
"""
from scrapy.http import TextResponse, XmlResponse
has_lxml = True
try:
from scrapy.selector.lxmlsel import XmlXPathSelector, HtmlXPathSelector, \
XPathSelector
except ImportError:
has_lxml = False
from scrapy.utils.test import libxml2debug
import unittest
from scrapy.tests import test_selector
from scrapy.http import TextResponse, HtmlResponse
from scrapy.selector.lxmldocument import LxmlDocument
from scrapy.selector.lxmlsel import XmlXPathSelector, HtmlXPathSelector, XPathSelector
class LxmlXPathSelectorTestCase(test_selector.XPathSelectorTestCase):
if has_lxml:
xs_cls = XPathSelector
hxs_cls = HtmlXPathSelector
xxs_cls = XmlXPathSelector
else:
skip = "lxml not available"
xs_cls = XPathSelector
hxs_cls = HtmlXPathSelector
xxs_cls = XmlXPathSelector
# XXX: this test was disabled because lxml behaves inconsistently when
# handling null bytes between different 2.2.x versions, but it may be due
# to differences in libxml2 too. it's also unclear what should be the
# proper behaviour (pablo - 26 oct 2010)
#@libxml2debug
#def test_null_bytes(self):
# hxs = HtmlXPathSelector(text='<root>la\x00la</root>')
# self.assertEqual(hxs.extract(),
# u'<html><body><root>la</root></body></html>')
#
# xxs = XmlXPathSelector(text='<root>la\x00la</root>')
# self.assertEqual(xxs.extract(),
# u'<root>la</root>')
class Libxml2DocumentTest(unittest.TestCase):
def test_caching(self):
r1 = HtmlResponse('http://www.example.com', body='<html><head></head><body></body></html>')
r2 = r1.copy()
doc1 = LxmlDocument(r1)
doc2 = LxmlDocument(r1)
doc3 = LxmlDocument(r2)
# make sure it's cached
assert doc1 is doc2
assert doc1 is not doc3
# don't leave documents in memory to avoid wrong libxml2 leaks reports
del doc1, doc2, doc3
def test_null_char(self):
# make sure bodies with null char ('\x00') don't raise a TypeError exception
self.body_content = 'test problematic \x00 body'
response = TextResponse('http://example.com/catalog/product/blabla-123',
headers={'Content-Type': 'text/plain; charset=utf-8'}, body=self.body_content)
LxmlDocument(response)