mirror of https://github.com/scrapy/scrapy.git
bumped embedded BeautifulSoup to 3.0.8.1
This commit is contained in:
parent
e528a77fa3
commit
02b7ca7e8c
|
|
@ -42,7 +42,7 @@ http://www.crummy.com/software/BeautifulSoup/documentation.html
|
|||
|
||||
Here, have some legalese:
|
||||
|
||||
Copyright (c) 2004-2008, Leonard Richardson
|
||||
Copyright (c) 2004-2010, Leonard Richardson
|
||||
|
||||
All rights reserved.
|
||||
|
||||
|
|
@ -79,8 +79,8 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE, DAMMIT.
|
|||
from __future__ import generators
|
||||
|
||||
__author__ = "Leonard Richardson (leonardr@segfault.org)"
|
||||
__version__ = "3.0.7a"
|
||||
__copyright__ = "Copyright (c) 2004-2008 Leonard Richardson"
|
||||
__version__ = "3.0.8.1"
|
||||
__copyright__ = "Copyright (c) 2004-2010 Leonard Richardson"
|
||||
__license__ = "New-style BSD"
|
||||
|
||||
from sgmllib import SGMLParser, SGMLParseError
|
||||
|
|
@ -104,9 +104,13 @@ markupbase._declname_match = re.compile(r'[a-zA-Z][-_.:a-zA-Z0-9]*\s*').match
|
|||
|
||||
DEFAULT_OUTPUT_ENCODING = "utf-8"
|
||||
|
||||
def _match_css_class(str):
|
||||
"""Build a RE to match the given CSS class."""
|
||||
return re.compile(r"(^|.*\s)%s($|\s)" % str)
|
||||
|
||||
# First, the classes that represent markup elements.
|
||||
|
||||
class PageElement:
|
||||
class PageElement(object):
|
||||
"""Contains the navigational information for some part of the page
|
||||
(either a tag or a piece of text)"""
|
||||
|
||||
|
|
@ -124,10 +128,11 @@ class PageElement:
|
|||
|
||||
def replaceWith(self, replaceWith):
|
||||
oldParent = self.parent
|
||||
myIndex = self.parent.contents.index(self)
|
||||
if hasattr(replaceWith, 'parent') and replaceWith.parent == self.parent:
|
||||
myIndex = self.parent.index(self)
|
||||
if hasattr(replaceWith, "parent")\
|
||||
and replaceWith.parent is self.parent:
|
||||
# We're replacing this element with one of its siblings.
|
||||
index = self.parent.contents.index(replaceWith)
|
||||
index = replaceWith.parent.index(replaceWith)
|
||||
if index and index < myIndex:
|
||||
# Furthermore, it comes before this element. That
|
||||
# means that when we extract it, the index of this
|
||||
|
|
@ -136,11 +141,20 @@ class PageElement:
|
|||
self.extract()
|
||||
oldParent.insert(myIndex, replaceWith)
|
||||
|
||||
def replaceWithChildren(self):
|
||||
myParent = self.parent
|
||||
myIndex = self.parent.index(self)
|
||||
self.extract()
|
||||
reversedChildren = list(self.contents)
|
||||
reversedChildren.reverse()
|
||||
for child in reversedChildren:
|
||||
myParent.insert(myIndex, child)
|
||||
|
||||
def extract(self):
|
||||
"""Destructively rips this element out of the tree."""
|
||||
if self.parent:
|
||||
try:
|
||||
self.parent.contents.remove(self)
|
||||
del self.parent.contents[self.parent.index(self)]
|
||||
except ValueError:
|
||||
pass
|
||||
|
||||
|
|
@ -173,18 +187,17 @@ class PageElement:
|
|||
return lastChild
|
||||
|
||||
def insert(self, position, newChild):
|
||||
if (isinstance(newChild, basestring)
|
||||
or isinstance(newChild, unicode)) \
|
||||
if isinstance(newChild, basestring) \
|
||||
and not isinstance(newChild, NavigableString):
|
||||
newChild = NavigableString(newChild)
|
||||
|
||||
position = min(position, len(self.contents))
|
||||
if hasattr(newChild, 'parent') and newChild.parent != None:
|
||||
if hasattr(newChild, 'parent') and newChild.parent is not None:
|
||||
# We're 'inserting' an element that's already one
|
||||
# of this object's children.
|
||||
if newChild.parent == self:
|
||||
index = self.find(newChild)
|
||||
if index and index < position:
|
||||
if newChild.parent is self:
|
||||
index = self.index(newChild)
|
||||
if index > position:
|
||||
# Furthermore we're moving it further down the
|
||||
# list of this object's children. That means that
|
||||
# when we extract this element, our target index
|
||||
|
|
@ -322,8 +335,21 @@ class PageElement:
|
|||
|
||||
if isinstance(name, SoupStrainer):
|
||||
strainer = name
|
||||
# (Possibly) special case some findAll*(...) searches
|
||||
elif text is None and not limit and not attrs and not kwargs:
|
||||
# findAll*(True)
|
||||
if name is True:
|
||||
return [element for element in generator()
|
||||
if isinstance(element, Tag)]
|
||||
# findAll*('tag-name')
|
||||
elif isinstance(name, basestring):
|
||||
return [element for element in generator()
|
||||
if isinstance(element, Tag) and
|
||||
element.name == name]
|
||||
else:
|
||||
strainer = SoupStrainer(name, attrs, text, **kwargs)
|
||||
# Build a SoupStrainer
|
||||
else:
|
||||
# Build a SoupStrainer
|
||||
strainer = SoupStrainer(name, attrs, text, **kwargs)
|
||||
results = ResultSet(strainer)
|
||||
g = generator()
|
||||
|
|
@ -344,31 +370,31 @@ class PageElement:
|
|||
#NavigableStrings and Tags.
|
||||
def nextGenerator(self):
|
||||
i = self
|
||||
while i:
|
||||
while i is not None:
|
||||
i = i.next
|
||||
yield i
|
||||
|
||||
def nextSiblingGenerator(self):
|
||||
i = self
|
||||
while i:
|
||||
while i is not None:
|
||||
i = i.nextSibling
|
||||
yield i
|
||||
|
||||
def previousGenerator(self):
|
||||
i = self
|
||||
while i:
|
||||
while i is not None:
|
||||
i = i.previous
|
||||
yield i
|
||||
|
||||
def previousSiblingGenerator(self):
|
||||
i = self
|
||||
while i:
|
||||
while i is not None:
|
||||
i = i.previousSibling
|
||||
yield i
|
||||
|
||||
def parentGenerator(self):
|
||||
i = self
|
||||
while i:
|
||||
while i is not None:
|
||||
i = i.parent
|
||||
yield i
|
||||
|
||||
|
|
@ -503,7 +529,7 @@ class Tag(PageElement):
|
|||
self.parserClass = parser.__class__
|
||||
self.isSelfClosing = parser.isSelfClosingTag(name)
|
||||
self.name = name
|
||||
if attrs == None:
|
||||
if attrs is None:
|
||||
attrs = []
|
||||
self.attrs = attrs
|
||||
self.contents = []
|
||||
|
|
@ -521,12 +547,49 @@ class Tag(PageElement):
|
|||
val))
|
||||
self.attrs = map(convert, self.attrs)
|
||||
|
||||
def getString(self):
|
||||
if (len(self.contents) == 1
|
||||
and isinstance(self.contents[0], NavigableString)):
|
||||
return self.contents[0]
|
||||
|
||||
def setString(self, string):
|
||||
"""Replace the contents of the tag with a string"""
|
||||
self.clear()
|
||||
self.append(string)
|
||||
|
||||
string = property(getString, setString)
|
||||
|
||||
def getText(self, separator=u""):
|
||||
if not len(self.contents):
|
||||
return u""
|
||||
stopNode = self._lastRecursiveChild().next
|
||||
strings = []
|
||||
current = self.contents[0]
|
||||
while current is not stopNode:
|
||||
if isinstance(current, NavigableString):
|
||||
strings.append(current.strip())
|
||||
current = current.next
|
||||
return separator.join(strings)
|
||||
|
||||
text = property(getText)
|
||||
|
||||
def get(self, key, default=None):
|
||||
"""Returns the value of the 'key' attribute for the tag, or
|
||||
the value given for 'default' if it doesn't have that
|
||||
attribute."""
|
||||
return self._getAttrMap().get(key, default)
|
||||
|
||||
def clear(self):
|
||||
"""Extract all children."""
|
||||
for child in self.contents[:]:
|
||||
child.extract()
|
||||
|
||||
def index(self, element):
|
||||
for i, child in enumerate(self.contents):
|
||||
if child is element:
|
||||
return i
|
||||
raise ValueError("Tag.index: element not in tag")
|
||||
|
||||
def has_key(self, key):
|
||||
return self._getAttrMap().has_key(key)
|
||||
|
||||
|
|
@ -595,6 +658,8 @@ class Tag(PageElement):
|
|||
|
||||
NOTE: right now this will return false if two tags have the
|
||||
same attributes in a different order. Should this be fixed?"""
|
||||
if other is self:
|
||||
return True
|
||||
if not hasattr(other, 'name') or not hasattr(other, 'attrs') or not hasattr(other, 'contents') or self.name != other.name or self.attrs != other.attrs or len(self) != len(other):
|
||||
return False
|
||||
for i in range(0, len(self.contents)):
|
||||
|
|
@ -638,7 +703,7 @@ class Tag(PageElement):
|
|||
if self.attrs:
|
||||
for key, val in self.attrs:
|
||||
fmt = '%s="%s"'
|
||||
if isString(val):
|
||||
if isinstance(val, basestring):
|
||||
if self.containsSubstitutions and '%SOUP-ENCODING%' in val:
|
||||
val = self.substituteEncoding(val, encoding)
|
||||
|
||||
|
|
@ -710,13 +775,20 @@ class Tag(PageElement):
|
|||
|
||||
def decompose(self):
|
||||
"""Recursively destroys the contents of this tree."""
|
||||
contents = [i for i in self.contents]
|
||||
for i in contents:
|
||||
if isinstance(i, Tag):
|
||||
i.decompose()
|
||||
else:
|
||||
i.extract()
|
||||
self.extract()
|
||||
if len(self.contents) == 0:
|
||||
return
|
||||
current = self.contents[0]
|
||||
while current is not None:
|
||||
next = current.next
|
||||
if isinstance(current, Tag):
|
||||
del current.contents[:]
|
||||
current.parent = None
|
||||
current.previous = None
|
||||
current.previousSibling = None
|
||||
current.next = None
|
||||
current.nextSibling = None
|
||||
current = next
|
||||
|
||||
def prettify(self, encoding=DEFAULT_OUTPUT_ENCODING):
|
||||
return self.__str__(encoding, True)
|
||||
|
|
@ -795,24 +867,18 @@ class Tag(PageElement):
|
|||
|
||||
#Generator methods
|
||||
def childGenerator(self):
|
||||
for i in range(0, len(self.contents)):
|
||||
yield self.contents[i]
|
||||
raise StopIteration
|
||||
# Just use the iterator from the contents
|
||||
return iter(self.contents)
|
||||
|
||||
def recursiveChildGenerator(self):
|
||||
stack = [(self, 0)]
|
||||
while stack:
|
||||
tag, start = stack.pop()
|
||||
if isinstance(tag, Tag):
|
||||
for i in range(start, len(tag.contents)):
|
||||
a = tag.contents[i]
|
||||
yield a
|
||||
if isinstance(a, Tag) and tag.contents:
|
||||
if i < len(tag.contents) - 1:
|
||||
stack.append((tag, i+1))
|
||||
stack.append((a, 0))
|
||||
break
|
||||
raise StopIteration
|
||||
if not len(self.contents):
|
||||
raise StopIteration
|
||||
stopNode = self._lastRecursiveChild().next
|
||||
current = self.contents[0]
|
||||
while current is not stopNode:
|
||||
yield current
|
||||
current = current.next
|
||||
|
||||
|
||||
# Next, a couple classes to represent queries and their results.
|
||||
class SoupStrainer:
|
||||
|
|
@ -821,8 +887,8 @@ class SoupStrainer:
|
|||
|
||||
def __init__(self, name=None, attrs={}, text=None, **kwargs):
|
||||
self.name = name
|
||||
if isString(attrs):
|
||||
kwargs['class'] = attrs
|
||||
if isinstance(attrs, basestring):
|
||||
kwargs['class'] = _match_css_class(attrs)
|
||||
attrs = None
|
||||
if kwargs:
|
||||
if attrs:
|
||||
|
|
@ -881,7 +947,8 @@ class SoupStrainer:
|
|||
found = None
|
||||
# If given a list of items, scan it for a text element that
|
||||
# matches.
|
||||
if isList(markup) and not isinstance(markup, Tag):
|
||||
if hasattr(markup, "__iter__") \
|
||||
and not isinstance(markup, Tag):
|
||||
for element in markup:
|
||||
if isinstance(element, NavigableString) \
|
||||
and self.search(element):
|
||||
|
|
@ -894,7 +961,7 @@ class SoupStrainer:
|
|||
found = self.searchTag(markup)
|
||||
# If it's text, make sure the text matches.
|
||||
elif isinstance(markup, NavigableString) or \
|
||||
isString(markup):
|
||||
isinstance(markup, basestring):
|
||||
if self._matches(markup, self.text):
|
||||
found = markup
|
||||
else:
|
||||
|
|
@ -905,8 +972,8 @@ class SoupStrainer:
|
|||
def _matches(self, markup, matchAgainst):
|
||||
#print "Matching %s against %s" % (markup, matchAgainst)
|
||||
result = False
|
||||
if matchAgainst == True and type(matchAgainst) == types.BooleanType:
|
||||
result = markup != None
|
||||
if matchAgainst is True:
|
||||
result = markup is not None
|
||||
elif callable(matchAgainst):
|
||||
result = matchAgainst(markup)
|
||||
else:
|
||||
|
|
@ -914,17 +981,17 @@ class SoupStrainer:
|
|||
#other ways of matching match the tag name as a string.
|
||||
if isinstance(markup, Tag):
|
||||
markup = markup.name
|
||||
if markup and not isString(markup):
|
||||
if markup and not isinstance(markup, basestring):
|
||||
markup = unicode(markup)
|
||||
#Now we know that chunk is either a string, or None.
|
||||
if hasattr(matchAgainst, 'match'):
|
||||
# It's a regexp object.
|
||||
result = markup and matchAgainst.search(markup)
|
||||
elif isList(matchAgainst):
|
||||
elif hasattr(matchAgainst, '__iter__'): # list-like
|
||||
result = markup in matchAgainst
|
||||
elif hasattr(matchAgainst, 'items'):
|
||||
result = markup.has_key(matchAgainst)
|
||||
elif matchAgainst and isString(markup):
|
||||
elif matchAgainst and isinstance(markup, basestring):
|
||||
if isinstance(markup, unicode):
|
||||
matchAgainst = unicode(matchAgainst)
|
||||
else:
|
||||
|
|
@ -943,20 +1010,6 @@ class ResultSet(list):
|
|||
|
||||
# Now, some helper functions.
|
||||
|
||||
def isList(l):
|
||||
"""Convenience method that works with all 2.x versions of Python
|
||||
to determine whether or not something is listlike."""
|
||||
return hasattr(l, '__iter__') \
|
||||
or (type(l) in (types.ListType, types.TupleType))
|
||||
|
||||
def isString(s):
|
||||
"""Convenience method that works with all 2.x versions of Python
|
||||
to determine whether or not something is stringlike."""
|
||||
try:
|
||||
return isinstance(s, unicode) or isinstance(s, basestring)
|
||||
except NameError:
|
||||
return isinstance(s, str)
|
||||
|
||||
def buildTagMap(default, *args):
|
||||
"""Turns a list of maps, lists, or scalars into a single map.
|
||||
Used to build the SELF_CLOSING_TAGS, NESTABLE_TAGS, and
|
||||
|
|
@ -967,7 +1020,7 @@ def buildTagMap(default, *args):
|
|||
#It's a map. Merge it.
|
||||
for k,v in portion.items():
|
||||
built[k] = v
|
||||
elif isList(portion):
|
||||
elif hasattr(portion, '__iter__'): # is a list
|
||||
#It's a list. Map each item to the default.
|
||||
for k in portion:
|
||||
built[k] = default
|
||||
|
|
@ -1116,7 +1169,7 @@ class BeautifulStoneSoup(Tag, SGMLParser):
|
|||
self.declaredHTMLEncoding = dammit.declaredHTMLEncoding
|
||||
if markup:
|
||||
if self.markupMassage:
|
||||
if not isList(self.markupMassage):
|
||||
if not hasattr(self.markupMassage, "__iter__"):
|
||||
self.markupMassage = self.MARKUP_MASSAGE
|
||||
for fix, m in self.markupMassage:
|
||||
markup = fix.sub(m, markup)
|
||||
|
|
@ -1139,10 +1192,10 @@ class BeautifulStoneSoup(Tag, SGMLParser):
|
|||
superclass or the Tag superclass, depending on the method name."""
|
||||
#print "__getattr__ called on %s.%s" % (self.__class__, methodName)
|
||||
|
||||
if methodName.find('start_') == 0 or methodName.find('end_') == 0 \
|
||||
or methodName.find('do_') == 0:
|
||||
if methodName.startswith('start_') or methodName.startswith('end_') \
|
||||
or methodName.startswith('do_'):
|
||||
return SGMLParser.__getattr__(self, methodName)
|
||||
elif methodName.find('__') != 0:
|
||||
elif not methodName.startswith('__'):
|
||||
return Tag.__getattr__(self, methodName)
|
||||
else:
|
||||
raise AttributeError
|
||||
|
|
@ -1165,12 +1218,6 @@ class BeautifulStoneSoup(Tag, SGMLParser):
|
|||
|
||||
def popTag(self):
|
||||
tag = self.tagStack.pop()
|
||||
# Tags with just one string-owning child get the child as a
|
||||
# 'string' property, so that soup.tag.string is shorthand for
|
||||
# soup.tag.contents[0]
|
||||
if len(self.currentTag.contents) == 1 and \
|
||||
isinstance(self.currentTag.contents[0], NavigableString):
|
||||
self.currentTag.string = self.currentTag.contents[0]
|
||||
|
||||
#print "Pop", tag.name
|
||||
if self.tagStack:
|
||||
|
|
@ -1259,9 +1306,9 @@ class BeautifulStoneSoup(Tag, SGMLParser):
|
|||
#last occurance.
|
||||
popTo = name
|
||||
break
|
||||
if (nestingResetTriggers != None
|
||||
if (nestingResetTriggers is not None
|
||||
and p.name in nestingResetTriggers) \
|
||||
or (nestingResetTriggers == None and isResetNesting
|
||||
or (nestingResetTriggers is None and isResetNesting
|
||||
and self.RESET_NESTING_TAGS.has_key(p.name)):
|
||||
|
||||
#If we encounter one of the nesting reset triggers
|
||||
|
|
@ -1280,7 +1327,7 @@ class BeautifulStoneSoup(Tag, SGMLParser):
|
|||
if self.quoteStack:
|
||||
#This is not a real tag.
|
||||
#print "<%s> is not real!" % name
|
||||
attrs = ''.join(map(lambda(x, y): ' %s="%s"' % (x, y), attrs))
|
||||
attrs = ''.join([' %s="%s"' % (x, y) for x, y in attrs])
|
||||
self.handle_data('<%s%s>' % (name, attrs))
|
||||
return
|
||||
self.endData()
|
||||
|
|
@ -1470,8 +1517,8 @@ class BeautifulSoup(BeautifulStoneSoup):
|
|||
BeautifulStoneSoup.__init__(self, *args, **kwargs)
|
||||
|
||||
SELF_CLOSING_TAGS = buildTagMap(None,
|
||||
['br' , 'hr', 'input', 'img', 'meta',
|
||||
'spacer', 'link', 'frame', 'base'])
|
||||
('br' , 'hr', 'input', 'img', 'meta',
|
||||
'spacer', 'link', 'frame', 'base', 'col'))
|
||||
|
||||
PRESERVE_WHITESPACE_TAGS = set(['pre', 'textarea'])
|
||||
|
||||
|
|
@ -1480,13 +1527,13 @@ class BeautifulSoup(BeautifulStoneSoup):
|
|||
#According to the HTML standard, each of these inline tags can
|
||||
#contain another tag of the same type. Furthermore, it's common
|
||||
#to actually use these tags this way.
|
||||
NESTABLE_INLINE_TAGS = ['span', 'font', 'q', 'object', 'bdo', 'sub', 'sup',
|
||||
'center']
|
||||
NESTABLE_INLINE_TAGS = ('span', 'font', 'q', 'object', 'bdo', 'sub', 'sup',
|
||||
'center')
|
||||
|
||||
#According to the HTML standard, these block tags can contain
|
||||
#another tag of the same type. Furthermore, it's common
|
||||
#to actually use these tags this way.
|
||||
NESTABLE_BLOCK_TAGS = ['blockquote', 'div', 'fieldset', 'ins', 'del']
|
||||
NESTABLE_BLOCK_TAGS = ('blockquote', 'div', 'fieldset', 'ins', 'del')
|
||||
|
||||
#Lists can contain other lists, but there are restrictions.
|
||||
NESTABLE_LIST_TAGS = { 'ol' : [],
|
||||
|
|
@ -1506,7 +1553,7 @@ class BeautifulSoup(BeautifulStoneSoup):
|
|||
'tfoot' : ['table'],
|
||||
}
|
||||
|
||||
NON_NESTABLE_BLOCK_TAGS = ['address', 'form', 'p', 'pre']
|
||||
NON_NESTABLE_BLOCK_TAGS = ('address', 'form', 'p', 'pre')
|
||||
|
||||
#If one of these tags is encountered, all tags up to the next tag of
|
||||
#this type are popped.
|
||||
|
|
@ -1597,11 +1644,11 @@ class ICantBelieveItsBeautifulSoup(BeautifulSoup):
|
|||
wouldn't be."""
|
||||
|
||||
I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS = \
|
||||
['em', 'big', 'i', 'small', 'tt', 'abbr', 'acronym', 'strong',
|
||||
('em', 'big', 'i', 'small', 'tt', 'abbr', 'acronym', 'strong',
|
||||
'cite', 'code', 'dfn', 'kbd', 'samp', 'strong', 'var', 'b',
|
||||
'big']
|
||||
'big')
|
||||
|
||||
I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS = ['noscript']
|
||||
I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS = ('noscript',)
|
||||
|
||||
NESTABLE_TAGS = buildTagMap([], BeautifulSoup.NESTABLE_TAGS,
|
||||
I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS,
|
||||
|
|
@ -1752,7 +1799,7 @@ class UnicodeDammit:
|
|||
"""Changes a MS smart quote character to an XML or HTML
|
||||
entity."""
|
||||
sub = self.MS_CHARS.get(orig)
|
||||
if type(sub) == types.TupleType:
|
||||
if isinstance(sub, tuple):
|
||||
if self.smartQuotesTo == 'xml':
|
||||
sub = '&#x%s;' % sub[1]
|
||||
else:
|
||||
|
|
|
|||
Loading…
Reference in New Issue