diff --git a/scrapy/contrib/ibl/htmlpage.py b/scrapy/contrib/ibl/htmlpage.py
index daeef4053..023030bdc 100644
--- a/scrapy/contrib/ibl/htmlpage.py
+++ b/scrapy/contrib/ibl/htmlpage.py
@@ -83,93 +83,18 @@ class HtmlTag(HtmlDataFragment):
_ATTR = "((?:[^=/>\s]|/(?!>))+)(?:\s*=(?:\s*\"(.*?)\"|\s*'(.*?)'|([^>\s]+))?)?"
_TAG = "<(\/?)(\w+(?::\w+)?)((?:\s+" + _ATTR + ")+\s*|\s*)(\/?)>"
_DOCTYPE = r""
+_SCRIPT = "()(.*?)()"
+_COMMENT = "()"
_ATTR_REGEXP = re.compile(_ATTR, re.I | re.DOTALL)
-_HTML_REGEXP = re.compile(_TAG, re.I | re.DOTALL)
+_HTML_REGEXP = re.compile("%s|%s|%s" % (_COMMENT, _SCRIPT, _TAG), re.I | re.DOTALL)
_DOCTYPE_REGEXP = re.compile("(?:%s)" % _DOCTYPE)
-_COMMENT_RE = re.compile("()", re.DOTALL)
-_SCRIPT_RE = re.compile("().*?()", re.DOTALL | re.I)
+_COMMENT_REGEXP = re.compile(_COMMENT, re.DOTALL)
def parse_html(text):
"""Higher level html parser. Calls lower level parsers and joins sucesive
HtmlDataFragment elements in a single one.
"""
- script_layer = lambda x: _parse_clean_html(x, _SCRIPT_RE, HtmlTag, _simple_parse_html)
- comment_layer = lambda x: _parse_clean_html(x, _COMMENT_RE, HtmlDataFragment, script_layer)
- delayed_element = None
- for element in comment_layer(text):
- if isinstance(element, HtmlTag):
- if delayed_element is not None:
- yield delayed_element
- delayed_element = None
- yield element
- else:# element is HtmlDataFragment
- if delayed_element is not None:
- delayed_element.start = min(element.start, delayed_element.start)
- delayed_element.end = max(element.end, delayed_element.end)
- else:
- delayed_element = element
- if delayed_element is not None:
- yield delayed_element
-
-def _parse_clean_html(text, regex, htype, func):
- """
- Removes regions from text, passes the cleaned text to the lower parse layer,
- and reinserts removed regions.
- regex - regular expression that defines regions to be removed/re inserted
- htype - the html parser type of the removed elements
- func - function that performs the lower parse layer
- """
- removed = [[m.start(), m.end(), m.groups()] for m in regex.finditer(text)]
-
- cleaned = regex.sub("", text)
- shift = 0
- for element in func(cleaned):
- element.start += shift
- element.end += shift
- while removed:
- if element.end <= removed[0][0]:
- yield element
- break
- else:
- start, end, groups = removed.pop(0)
- add = end - start
- element.end += add
- shift += add
- if element.start >= start:
- element.start += add
- elif isinstance(element, HtmlTag):
- yield element
- break
-
- if element.start < start:
- yield HtmlDataFragment(element.start, start)
- element.start = end
-
- if htype == HtmlTag:
- begintag = _parse_tag(_HTML_REGEXP.match(groups[0]))
- endtag = _parse_tag(_HTML_REGEXP.match(groups[1]))
- begintag.start = start
- begintag.end += start
-
- endtag.start = end - endtag.end
- endtag.end = end
- content = None
- if begintag.end < endtag.start:
- content = HtmlDataFragment(begintag.end, endtag.start)
- yield begintag
- if content is not None:
- yield content
- yield endtag
- else:
- yield htype(start, end)
- else:
- yield element
-
-def _simple_parse_html(text):
- """Simple html parse. It returns a sequence of HtmlTag and HtmlDataFragment
- objects. Does not ignore any region.
- """
# If have doctype remove it.
start_pos = 0
match = _DOCTYPE_REGEXP.match(text)
@@ -182,19 +107,49 @@ def _simple_parse_html(text):
if start > prev_end:
yield HtmlDataFragment(prev_end, start)
-
- yield _parse_tag(match)
+
+ if match.groups()[0] is not None: # comment
+ yield HtmlDataFragment(start, end)
+ elif match.groups()[1] is not None: #
+ for e in _parse_script(match):
+ yield e
+ else: # tag
+ yield _parse_tag(match)
prev_end = end
textlen = len(text)
if prev_end < textlen:
yield HtmlDataFragment(prev_end, textlen)
+def _parse_script(match):
+ """parse a region matched by _HTML_REGEXP"""
+ open_text, content, close_text = match.groups()[1:4]
+
+ open_tag = _parse_tag(_HTML_REGEXP.match(open_text))
+ open_tag.start = match.start()
+ open_tag.end = match.start() + len(open_text)
+
+ close_tag = _parse_tag(_HTML_REGEXP.match(close_text))
+ close_tag.start = match.end() - len(close_text)
+ close_tag.end = match.end()
+
+ yield open_tag
+ if open_tag.end < close_tag.start:
+ start_pos = 0
+ for m in _COMMENT_REGEXP.finditer(content):
+ if m.start() > start_pos:
+ yield HtmlDataFragment(open_tag.end + start_pos, open_tag.end + m.start())
+ yield HtmlDataFragment(open_tag.end + m.start(), open_tag.end + m.end())
+ start_pos = m.end()
+ if open_tag.end + start_pos < close_tag.start:
+ yield HtmlDataFragment(open_tag.end + start_pos, close_tag.start)
+ yield close_tag
+
def _parse_tag(match):
"""
parse a tag matched by _HTML_REGEXP
"""
data = match.groups()
- closing, tag, attr_text = data[:3]
+ closing, tag, attr_text = data[4:7]
# if tag is None then the match is a comment
if tag is not None:
unpaired = data[-1]
diff --git a/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_0.json b/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_0.json
index 13fb0d730..08826aa69 100644
--- a/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_0.json
+++ b/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_0.json
@@ -208,11 +208,19 @@
"end": 1073,
"start": 1043,
"tag_type": 1
- },
+ },
+ {
+ "start": 1073,
+ "end": 1074
+ },
{
- "start": 1073,
+ "start": 1074,
+ "end": 2052
+ },
+ {
+ "start": 2052,
"end": 2053
- },
+ },
{
"attributes": {},
"tag": "script",
@@ -3088,4 +3096,4 @@
"start": 13843,
"tag_type": 2
}
-]
\ No newline at end of file
+]
diff --git a/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_1.json b/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_1.json
index 15e661399..f94837c4d 100644
--- a/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_1.json
+++ b/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_1.json
@@ -208,8 +208,16 @@
},
{
"start": 1954,
+ "end": 1956
+ },
+ {
+ "start": 1956,
+ "end": 1979
+ },
+ {
+ "start": 1979,
"end": 1980
- },
+ },
{
"attributes": {
"src": "http://images.play.com/sitetrak/cmdatatagutilsA.js",
@@ -266,6 +274,14 @@
},
{
"start": 2282,
+ "end": 2283
+ },
+ {
+ "start": 2283,
+ "end": 2437
+ },
+ {
+ "start": 2437,
"end": 2438
},
{
@@ -325,6 +341,14 @@
},
{
"start": 2860,
+ "end": 2861
+ },
+ {
+ "start": 2861,
+ "end": 2882
+ },
+ {
+ "start": 2882,
"end": 2884
},
{
@@ -21941,4 +21965,4 @@
"tag_type": 2,
"start": 72792
}
-]
\ No newline at end of file
+]
diff --git a/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_2.json b/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_2.json
index 3fa134f28..c856f4a92 100644
--- a/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_2.json
+++ b/scrapy/tests/test_contrib_ibl/samples/samples_htmlpage_2.json
@@ -240,6 +240,14 @@
},
{
"start": 2180,
+ "end": 2182
+ },
+ {
+ "start": 2182,
+ "end": 2205
+ },
+ {
+ "start": 2205,
"end": 2206
},
{
@@ -298,6 +306,14 @@
},
{
"start": 2508,
+ "end": 2509
+ },
+ {
+ "start": 2509,
+ "end": 2663
+ },
+ {
+ "start": 2663,
"end": 2664
},
{
@@ -356,7 +372,15 @@
"tag_type": 2
},
{
- "start": 3086,
+ "start": 3086,
+ "end": 3087
+ },
+ {
+ "start": 3087,
+ "end": 3108
+ },
+ {
+ "start": 3108,
"end": 3110
},
{
@@ -21765,4 +21789,4 @@
"start": 72321,
"end": 72325
}
-]
\ No newline at end of file
+]
diff --git a/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py b/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py
index 53992fef1..62cd6b526 100644
--- a/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py
+++ b/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py
@@ -165,7 +165,9 @@ PARSED3 = [
{'attributes': {}, 'end': 55, 'start': 51, 'tag': u'p', 'tag_type': 2},
{'end': 70, 'start': 55},
{'attributes': {u'type': u'text/javascript'}, 'end': 101, 'start': 70, 'tag': u'script', 'tag_type': 1},
- {'end': 124, 'start': 101},
+ {'end': 104, 'start': 101},
+ {'end': 118, 'start': 104},
+ {'end': 124, 'start': 118},
{'attributes': {}, 'end': 133, 'start': 124, 'tag': u'script', 'tag_type': 2},
{'attributes': {}, 'end': 140, 'start': 133, 'tag': u'body', 'tag_type': 2},
{'attributes': {}, 'end': 147, 'start': 140, 'tag': u'html', 'tag_type': 2}
@@ -204,7 +206,8 @@ PARSED5 = [
{'end': 45, 'start': 42},
{'attributes': {}, 'end': 54, 'start': 45, 'tag': u'script', 'tag_type': 2},
{'attributes': {}, 'end': 61, 'start': 54, 'tag': u'body', 'tag_type': 2},
- {'end': 91, 'start': 61},
+ {'end': 76, 'start': 61},
+ {'end': 91, 'start': 76},
{'attributes': {}, 'end': 98, 'start': 91, 'tag': u'html', 'tag_type': 2},
]
@@ -215,7 +218,9 @@ PARSED6 = [
{'attributes': {}, 'end': 6, 'start': 0, 'tag': u'html', 'tag_type': 1},
{'attributes': {}, 'end': 12, 'start': 6, 'tag': u'body', 'tag_type': 1},
{'attributes': {}, 'end': 20, 'start': 12, 'tag': u'script', 'tag_type': 1},
- {'end': 40, 'start': 20},
+ {'end': 23, 'start': 20},
+ {'end': 37, 'start': 23},
+ {'end': 40, 'start': 37},
{'attributes': {}, 'end': 49, 'start': 40, 'tag': u'script', 'tag_type': 2},
{'end': 52, 'start': 49},
{'attributes': {}, 'end': 60, 'start': 52, 'tag': u'script', 'tag_type': 1},
@@ -225,3 +230,19 @@ PARSED6 = [
{'attributes': {}, 'end': 81, 'start': 74, 'tag': u'body', 'tag_type': 2},
{'attributes': {}, 'end': 88, 'start': 81, 'tag': u'html', 'tag_type': 2},
]
+
+# Test source without ending body nor html
+PAGE7 = u"""
veris in temporibus sub aprilis idibus
"""
+
+PARSED7 = [
+ {'attributes' : {}, 'end': 6, 'start': 0, 'tag': u'html', 'tag_type': 1},
+ {'attributes': {}, 'end': 12, 'start': 6, 'tag': u'body', 'tag_type': 1},
+ {'attributes': {}, 'end': 15, 'start': 12, 'tag': u'p', 'tag_type': 1},
+ {'end': 53, 'start': 15},
+ {'attributes': {}, 'end': 57, 'start': 53, 'tag': u'p', 'tag_type': 2},
+ {'attributes' : {}, 'end': 65, 'start': 57, 'tag': u'script', 'tag_type': 1},
+ {'end': 76, 'start': 65},
+ {'attributes' : {}, 'end': 85, 'start': 76, 'tag': u'script', 'tag_type': 2},
+ {'end': 99, 'start': 85},
+]
+