Removed some bogus decoding of strings in several markup-handling functions. Input must always be passed in a unicode object

--HG--
extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40543
This commit is contained in:
elpolilla 2008-12-23 23:12:56 +00:00
parent 6d33e12c8b
commit f4cecafc02
1 changed files with 12 additions and 12 deletions

View File

@ -11,7 +11,7 @@ _tag_re = re.compile(r'<[a-zA-Z\/!].*?>', re.DOTALL)
def remove_entities(text, keep=(), remove_illegal=True):
"""Remove entities from the given text.
'text' can be a unicode string or a regular string encoded as 'utf-8'
'text' must be a unicode string.
If 'keep' is passed (with a list of entity names) those entities will
be kept (they won't be removed).
@ -45,7 +45,7 @@ def remove_entities(text, keep=(), remove_illegal=True):
else:
return u'&%s;' % m.group(2)
return _ent_re.sub(convert_entity, text.decode('utf-8'))
return _ent_re.sub(convert_entity, text)
def has_entities(text):
return bool(_ent_re.search(text))
@ -54,16 +54,16 @@ def replace_tags(text, token=''):
"""Replace all markup tags found in the given text by the given token. By
default token is a null string so it just remove all tags.
'text' can be a unicode string or a regular string encoded as 'utf-8'
'text' must be a unicode string.
Always returns a unicode string.
"""
return _tag_re.sub(token, text.decode('utf-8'))
return _tag_re.sub(token, text)
def remove_comments(text):
""" Remove HTML Comments. """
return re.sub('<!--.*?-->', u'', text.decode('utf-8'), re.DOTALL)
return re.sub('<!--.*?-->', u'', text, re.DOTALL)
def remove_tags(text, which_ones=()):
""" Remove HTML Tags only.
@ -77,7 +77,7 @@ def remove_tags(text, which_ones=()):
else:
reg_exp_remove_tags = '<.*?>'
re_tags = re.compile(reg_exp_remove_tags, re.DOTALL)
return re_tags.sub(u'', text.decode('utf-8'))
return re_tags.sub(u'', text)
def remove_tags_with_content(text, which_ones=()):
""" Remove tags and its content.
@ -87,7 +87,7 @@ def remove_tags_with_content(text, which_ones=()):
"""
tags = [ '<%s.*?</%s>' % (tag,tag) for tag in which_ones ]
re_tags_remove = re.compile('|'.join(tags), re.DOTALL)
return re_tags_remove.sub(u'', text.decode('utf-8'))
return re_tags_remove.sub(u'', text)
def remove_escape_chars(text, which_ones=('\n','\t','\r')):
""" Remove escape chars. Default : \\n, \\t, \\r
@ -96,11 +96,11 @@ def remove_escape_chars(text, which_ones=('\n','\t','\r')):
By default removes \n, \t, \r.
"""
re_escape_chars = re.compile('[%s]' % ''.join(which_ones))
return re_escape_chars.sub(u'', text.decode('utf-8'))
return re_escape_chars.sub(u'', text)
def unquote_markup(text, keep=(), remove_illegal=True):
"""
This function receives markup as a text and does the following:
This function receives markup as a text (always a unicode string) and does the following:
- removes entities (except the ones in 'keep') from any part of it that it's not inside a CDATA
- searches for CDATAs and extracts their text (if any) without modifying it.
- removes the found CDATAs
@ -118,12 +118,12 @@ def unquote_markup(text, keep=(), remove_illegal=True):
fragments.append(txt[offset:])
return fragments
ret_text = ''
for fragment in _get_fragments(text.decode('utf-8'), _cdata_re):
ret_text = u''
for fragment in _get_fragments(text, _cdata_re):
if isinstance(fragment, basestring):
# it's not a CDATA (so we try to remove its entities)
ret_text += remove_entities(fragment, keep=keep, remove_illegal=remove_illegal)
else:
# it's a CDATA (so we just extract its content)
ret_text += fragment.group('cdata_d')
return unicode(ret_text)
return ret_text