diff --git a/scrapy/xlib/BeautifulSoup.py b/scrapy/xlib/BeautifulSoup.py index 0e214630c..748e6fe4b 100644 --- a/scrapy/xlib/BeautifulSoup.py +++ b/scrapy/xlib/BeautifulSoup.py @@ -42,7 +42,7 @@ http://www.crummy.com/software/BeautifulSoup/documentation.html Here, have some legalese: -Copyright (c) 2004-2008, Leonard Richardson +Copyright (c) 2004-2010, Leonard Richardson All rights reserved. @@ -79,8 +79,8 @@ SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE, DAMMIT. from __future__ import generators __author__ = "Leonard Richardson (leonardr@segfault.org)" -__version__ = "3.0.7a" -__copyright__ = "Copyright (c) 2004-2008 Leonard Richardson" +__version__ = "3.0.8.1" +__copyright__ = "Copyright (c) 2004-2010 Leonard Richardson" __license__ = "New-style BSD" from sgmllib import SGMLParser, SGMLParseError @@ -104,9 +104,13 @@ markupbase._declname_match = re.compile(r'[a-zA-Z][-_.:a-zA-Z0-9]*\s*').match DEFAULT_OUTPUT_ENCODING = "utf-8" +def _match_css_class(str): + """Build a RE to match the given CSS class.""" + return re.compile(r"(^|.*\s)%s($|\s)" % str) + # First, the classes that represent markup elements. -class PageElement: +class PageElement(object): """Contains the navigational information for some part of the page (either a tag or a piece of text)""" @@ -124,10 +128,11 @@ class PageElement: def replaceWith(self, replaceWith): oldParent = self.parent - myIndex = self.parent.contents.index(self) - if hasattr(replaceWith, 'parent') and replaceWith.parent == self.parent: + myIndex = self.parent.index(self) + if hasattr(replaceWith, "parent")\ + and replaceWith.parent is self.parent: # We're replacing this element with one of its siblings. - index = self.parent.contents.index(replaceWith) + index = replaceWith.parent.index(replaceWith) if index and index < myIndex: # Furthermore, it comes before this element. That # means that when we extract it, the index of this @@ -136,11 +141,20 @@ class PageElement: self.extract() oldParent.insert(myIndex, replaceWith) + def replaceWithChildren(self): + myParent = self.parent + myIndex = self.parent.index(self) + self.extract() + reversedChildren = list(self.contents) + reversedChildren.reverse() + for child in reversedChildren: + myParent.insert(myIndex, child) + def extract(self): """Destructively rips this element out of the tree.""" if self.parent: try: - self.parent.contents.remove(self) + del self.parent.contents[self.parent.index(self)] except ValueError: pass @@ -173,18 +187,17 @@ class PageElement: return lastChild def insert(self, position, newChild): - if (isinstance(newChild, basestring) - or isinstance(newChild, unicode)) \ + if isinstance(newChild, basestring) \ and not isinstance(newChild, NavigableString): newChild = NavigableString(newChild) position = min(position, len(self.contents)) - if hasattr(newChild, 'parent') and newChild.parent != None: + if hasattr(newChild, 'parent') and newChild.parent is not None: # We're 'inserting' an element that's already one # of this object's children. - if newChild.parent == self: - index = self.find(newChild) - if index and index < position: + if newChild.parent is self: + index = self.index(newChild) + if index > position: # Furthermore we're moving it further down the # list of this object's children. That means that # when we extract this element, our target index @@ -322,8 +335,21 @@ class PageElement: if isinstance(name, SoupStrainer): strainer = name + # (Possibly) special case some findAll*(...) searches + elif text is None and not limit and not attrs and not kwargs: + # findAll*(True) + if name is True: + return [element for element in generator() + if isinstance(element, Tag)] + # findAll*('tag-name') + elif isinstance(name, basestring): + return [element for element in generator() + if isinstance(element, Tag) and + element.name == name] + else: + strainer = SoupStrainer(name, attrs, text, **kwargs) + # Build a SoupStrainer else: - # Build a SoupStrainer strainer = SoupStrainer(name, attrs, text, **kwargs) results = ResultSet(strainer) g = generator() @@ -344,31 +370,31 @@ class PageElement: #NavigableStrings and Tags. def nextGenerator(self): i = self - while i: + while i is not None: i = i.next yield i def nextSiblingGenerator(self): i = self - while i: + while i is not None: i = i.nextSibling yield i def previousGenerator(self): i = self - while i: + while i is not None: i = i.previous yield i def previousSiblingGenerator(self): i = self - while i: + while i is not None: i = i.previousSibling yield i def parentGenerator(self): i = self - while i: + while i is not None: i = i.parent yield i @@ -503,7 +529,7 @@ class Tag(PageElement): self.parserClass = parser.__class__ self.isSelfClosing = parser.isSelfClosingTag(name) self.name = name - if attrs == None: + if attrs is None: attrs = [] self.attrs = attrs self.contents = [] @@ -521,12 +547,49 @@ class Tag(PageElement): val)) self.attrs = map(convert, self.attrs) + def getString(self): + if (len(self.contents) == 1 + and isinstance(self.contents[0], NavigableString)): + return self.contents[0] + + def setString(self, string): + """Replace the contents of the tag with a string""" + self.clear() + self.append(string) + + string = property(getString, setString) + + def getText(self, separator=u""): + if not len(self.contents): + return u"" + stopNode = self._lastRecursiveChild().next + strings = [] + current = self.contents[0] + while current is not stopNode: + if isinstance(current, NavigableString): + strings.append(current.strip()) + current = current.next + return separator.join(strings) + + text = property(getText) + def get(self, key, default=None): """Returns the value of the 'key' attribute for the tag, or the value given for 'default' if it doesn't have that attribute.""" return self._getAttrMap().get(key, default) + def clear(self): + """Extract all children.""" + for child in self.contents[:]: + child.extract() + + def index(self, element): + for i, child in enumerate(self.contents): + if child is element: + return i + raise ValueError("Tag.index: element not in tag") + def has_key(self, key): return self._getAttrMap().has_key(key) @@ -595,6 +658,8 @@ class Tag(PageElement): NOTE: right now this will return false if two tags have the same attributes in a different order. Should this be fixed?""" + if other is self: + return True if not hasattr(other, 'name') or not hasattr(other, 'attrs') or not hasattr(other, 'contents') or self.name != other.name or self.attrs != other.attrs or len(self) != len(other): return False for i in range(0, len(self.contents)): @@ -638,7 +703,7 @@ class Tag(PageElement): if self.attrs: for key, val in self.attrs: fmt = '%s="%s"' - if isString(val): + if isinstance(val, basestring): if self.containsSubstitutions and '%SOUP-ENCODING%' in val: val = self.substituteEncoding(val, encoding) @@ -710,13 +775,20 @@ class Tag(PageElement): def decompose(self): """Recursively destroys the contents of this tree.""" - contents = [i for i in self.contents] - for i in contents: - if isinstance(i, Tag): - i.decompose() - else: - i.extract() self.extract() + if len(self.contents) == 0: + return + current = self.contents[0] + while current is not None: + next = current.next + if isinstance(current, Tag): + del current.contents[:] + current.parent = None + current.previous = None + current.previousSibling = None + current.next = None + current.nextSibling = None + current = next def prettify(self, encoding=DEFAULT_OUTPUT_ENCODING): return self.__str__(encoding, True) @@ -795,24 +867,18 @@ class Tag(PageElement): #Generator methods def childGenerator(self): - for i in range(0, len(self.contents)): - yield self.contents[i] - raise StopIteration + # Just use the iterator from the contents + return iter(self.contents) def recursiveChildGenerator(self): - stack = [(self, 0)] - while stack: - tag, start = stack.pop() - if isinstance(tag, Tag): - for i in range(start, len(tag.contents)): - a = tag.contents[i] - yield a - if isinstance(a, Tag) and tag.contents: - if i < len(tag.contents) - 1: - stack.append((tag, i+1)) - stack.append((a, 0)) - break - raise StopIteration + if not len(self.contents): + raise StopIteration + stopNode = self._lastRecursiveChild().next + current = self.contents[0] + while current is not stopNode: + yield current + current = current.next + # Next, a couple classes to represent queries and their results. class SoupStrainer: @@ -821,8 +887,8 @@ class SoupStrainer: def __init__(self, name=None, attrs={}, text=None, **kwargs): self.name = name - if isString(attrs): - kwargs['class'] = attrs + if isinstance(attrs, basestring): + kwargs['class'] = _match_css_class(attrs) attrs = None if kwargs: if attrs: @@ -881,7 +947,8 @@ class SoupStrainer: found = None # If given a list of items, scan it for a text element that # matches. - if isList(markup) and not isinstance(markup, Tag): + if hasattr(markup, "__iter__") \ + and not isinstance(markup, Tag): for element in markup: if isinstance(element, NavigableString) \ and self.search(element): @@ -894,7 +961,7 @@ class SoupStrainer: found = self.searchTag(markup) # If it's text, make sure the text matches. elif isinstance(markup, NavigableString) or \ - isString(markup): + isinstance(markup, basestring): if self._matches(markup, self.text): found = markup else: @@ -905,8 +972,8 @@ class SoupStrainer: def _matches(self, markup, matchAgainst): #print "Matching %s against %s" % (markup, matchAgainst) result = False - if matchAgainst == True and type(matchAgainst) == types.BooleanType: - result = markup != None + if matchAgainst is True: + result = markup is not None elif callable(matchAgainst): result = matchAgainst(markup) else: @@ -914,17 +981,17 @@ class SoupStrainer: #other ways of matching match the tag name as a string. if isinstance(markup, Tag): markup = markup.name - if markup and not isString(markup): + if markup and not isinstance(markup, basestring): markup = unicode(markup) #Now we know that chunk is either a string, or None. if hasattr(matchAgainst, 'match'): # It's a regexp object. result = markup and matchAgainst.search(markup) - elif isList(matchAgainst): + elif hasattr(matchAgainst, '__iter__'): # list-like result = markup in matchAgainst elif hasattr(matchAgainst, 'items'): result = markup.has_key(matchAgainst) - elif matchAgainst and isString(markup): + elif matchAgainst and isinstance(markup, basestring): if isinstance(markup, unicode): matchAgainst = unicode(matchAgainst) else: @@ -943,20 +1010,6 @@ class ResultSet(list): # Now, some helper functions. -def isList(l): - """Convenience method that works with all 2.x versions of Python - to determine whether or not something is listlike.""" - return hasattr(l, '__iter__') \ - or (type(l) in (types.ListType, types.TupleType)) - -def isString(s): - """Convenience method that works with all 2.x versions of Python - to determine whether or not something is stringlike.""" - try: - return isinstance(s, unicode) or isinstance(s, basestring) - except NameError: - return isinstance(s, str) - def buildTagMap(default, *args): """Turns a list of maps, lists, or scalars into a single map. Used to build the SELF_CLOSING_TAGS, NESTABLE_TAGS, and @@ -967,7 +1020,7 @@ def buildTagMap(default, *args): #It's a map. Merge it. for k,v in portion.items(): built[k] = v - elif isList(portion): + elif hasattr(portion, '__iter__'): # is a list #It's a list. Map each item to the default. for k in portion: built[k] = default @@ -1116,7 +1169,7 @@ class BeautifulStoneSoup(Tag, SGMLParser): self.declaredHTMLEncoding = dammit.declaredHTMLEncoding if markup: if self.markupMassage: - if not isList(self.markupMassage): + if not hasattr(self.markupMassage, "__iter__"): self.markupMassage = self.MARKUP_MASSAGE for fix, m in self.markupMassage: markup = fix.sub(m, markup) @@ -1139,10 +1192,10 @@ class BeautifulStoneSoup(Tag, SGMLParser): superclass or the Tag superclass, depending on the method name.""" #print "__getattr__ called on %s.%s" % (self.__class__, methodName) - if methodName.find('start_') == 0 or methodName.find('end_') == 0 \ - or methodName.find('do_') == 0: + if methodName.startswith('start_') or methodName.startswith('end_') \ + or methodName.startswith('do_'): return SGMLParser.__getattr__(self, methodName) - elif methodName.find('__') != 0: + elif not methodName.startswith('__'): return Tag.__getattr__(self, methodName) else: raise AttributeError @@ -1165,12 +1218,6 @@ class BeautifulStoneSoup(Tag, SGMLParser): def popTag(self): tag = self.tagStack.pop() - # Tags with just one string-owning child get the child as a - # 'string' property, so that soup.tag.string is shorthand for - # soup.tag.contents[0] - if len(self.currentTag.contents) == 1 and \ - isinstance(self.currentTag.contents[0], NavigableString): - self.currentTag.string = self.currentTag.contents[0] #print "Pop", tag.name if self.tagStack: @@ -1259,9 +1306,9 @@ class BeautifulStoneSoup(Tag, SGMLParser): #last occurance. popTo = name break - if (nestingResetTriggers != None + if (nestingResetTriggers is not None and p.name in nestingResetTriggers) \ - or (nestingResetTriggers == None and isResetNesting + or (nestingResetTriggers is None and isResetNesting and self.RESET_NESTING_TAGS.has_key(p.name)): #If we encounter one of the nesting reset triggers @@ -1280,7 +1327,7 @@ class BeautifulStoneSoup(Tag, SGMLParser): if self.quoteStack: #This is not a real tag. #print "<%s> is not real!" % name - attrs = ''.join(map(lambda(x, y): ' %s="%s"' % (x, y), attrs)) + attrs = ''.join([' %s="%s"' % (x, y) for x, y in attrs]) self.handle_data('<%s%s>' % (name, attrs)) return self.endData() @@ -1470,8 +1517,8 @@ class BeautifulSoup(BeautifulStoneSoup): BeautifulStoneSoup.__init__(self, *args, **kwargs) SELF_CLOSING_TAGS = buildTagMap(None, - ['br' , 'hr', 'input', 'img', 'meta', - 'spacer', 'link', 'frame', 'base']) + ('br' , 'hr', 'input', 'img', 'meta', + 'spacer', 'link', 'frame', 'base', 'col')) PRESERVE_WHITESPACE_TAGS = set(['pre', 'textarea']) @@ -1480,13 +1527,13 @@ class BeautifulSoup(BeautifulStoneSoup): #According to the HTML standard, each of these inline tags can #contain another tag of the same type. Furthermore, it's common #to actually use these tags this way. - NESTABLE_INLINE_TAGS = ['span', 'font', 'q', 'object', 'bdo', 'sub', 'sup', - 'center'] + NESTABLE_INLINE_TAGS = ('span', 'font', 'q', 'object', 'bdo', 'sub', 'sup', + 'center') #According to the HTML standard, these block tags can contain #another tag of the same type. Furthermore, it's common #to actually use these tags this way. - NESTABLE_BLOCK_TAGS = ['blockquote', 'div', 'fieldset', 'ins', 'del'] + NESTABLE_BLOCK_TAGS = ('blockquote', 'div', 'fieldset', 'ins', 'del') #Lists can contain other lists, but there are restrictions. NESTABLE_LIST_TAGS = { 'ol' : [], @@ -1506,7 +1553,7 @@ class BeautifulSoup(BeautifulStoneSoup): 'tfoot' : ['table'], } - NON_NESTABLE_BLOCK_TAGS = ['address', 'form', 'p', 'pre'] + NON_NESTABLE_BLOCK_TAGS = ('address', 'form', 'p', 'pre') #If one of these tags is encountered, all tags up to the next tag of #this type are popped. @@ -1597,11 +1644,11 @@ class ICantBelieveItsBeautifulSoup(BeautifulSoup): wouldn't be.""" I_CANT_BELIEVE_THEYRE_NESTABLE_INLINE_TAGS = \ - ['em', 'big', 'i', 'small', 'tt', 'abbr', 'acronym', 'strong', + ('em', 'big', 'i', 'small', 'tt', 'abbr', 'acronym', 'strong', 'cite', 'code', 'dfn', 'kbd', 'samp', 'strong', 'var', 'b', - 'big'] + 'big') - I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS = ['noscript'] + I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS = ('noscript',) NESTABLE_TAGS = buildTagMap([], BeautifulSoup.NESTABLE_TAGS, I_CANT_BELIEVE_THEYRE_NESTABLE_BLOCK_TAGS, @@ -1752,7 +1799,7 @@ class UnicodeDammit: """Changes a MS smart quote character to an XML or HTML entity.""" sub = self.MS_CHARS.get(orig) - if type(sub) == types.TupleType: + if isinstance(sub, tuple): if self.smartQuotesTo == 'xml': sub = '&#x%s;' % sub[1] else: