diff --git a/docs/topics/scrapyd.rst b/docs/topics/scrapyd.rst index 6bfe8d3f7..143e983a3 100644 --- a/docs/topics/scrapyd.rst +++ b/docs/topics/scrapyd.rst @@ -86,7 +86,7 @@ in your Ubuntu servers. So, if you plan to deploy Scrapyd on a Ubuntu server, just add the Ubuntu repositories as described in :ref:`topics-ubuntu` and then run:: - aptitude install scrapyd-0.12 + aptitude install scrapyd-0.13 This will install Scrapyd in your Ubuntu server creating a ``scrapy`` user which Scrapyd will run as. It will also create some directories and files that diff --git a/docs/topics/ubuntu.rst b/docs/topics/ubuntu.rst index 13bdd4b7b..6cd164f7b 100644 --- a/docs/topics/ubuntu.rst +++ b/docs/topics/ubuntu.rst @@ -13,7 +13,7 @@ latest bug fixes. To use the packages, just add the following line to your ``/etc/apt/sources.list``, and then run ``aptitude update`` and ``aptitude -install scrapy-0.12``:: +install scrapy-0.13``:: deb http://archive.scrapy.org/ubuntu DISTRO main diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 3d8a7a4ae..c73dae498 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -2,8 +2,8 @@ Scrapy - a screen scraping framework written in Python """ -version_info = (0, 12, 0) -__version__ = "0.12.0" +version_info = (0, 13, 0) +__version__ = "0.13.0" import sys, os, warnings diff --git a/scrapy/contrib/ibl/htmlpage.py b/scrapy/contrib/ibl/htmlpage.py index 023030bdc..c86ec9670 100644 --- a/scrapy/contrib/ibl/htmlpage.py +++ b/scrapy/contrib/ibl/htmlpage.py @@ -80,8 +80,8 @@ class HtmlTag(HtmlDataFragment): def __repr__(self): return str(self) -_ATTR = "((?:[^=/>\s]|/(?!>))+)(?:\s*=(?:\s*\"(.*?)\"|\s*'(.*?)'|([^>\s]+))?)?" -_TAG = "<(\/?)(\w+(?::\w+)?)((?:\s+" + _ATTR + ")+\s*|\s*)(\/?)>" +_ATTR = "((?:[^=/<>\s]|/(?!>))+)(?:\s*=(?:\s*\"(.*?)\"|\s*'(.*?)'|([^>\s]+))?)?" +_TAG = "<(\/?)(\w+(?::\w+)?)((?:\s*" + _ATTR + ")+\s*|\s*)(\/?)>?" _DOCTYPE = r"" _SCRIPT = "()(.*?)()" _COMMENT = "()" diff --git a/scrapy/link.py b/scrapy/link.py index ddae07823..371a987cb 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -18,6 +18,9 @@ class Link(object): def __eq__(self, other): return self.url == other.url and self.text == other.text + + def __hash__(self): + return hash(self.url) ^ hash(self.text) def __repr__(self): return '' % (self.url, self.text) diff --git a/scrapy/tests/test_contrib_ibl/test_extraction.py b/scrapy/tests/test_contrib_ibl/test_extraction.py index f79c9dc78..e9ce86f7e 100644 --- a/scrapy/tests/test_contrib_ibl/test_extraction.py +++ b/scrapy/tests/test_contrib_ibl/test_extraction.py @@ -516,7 +516,7 @@ ANNOTATED_PAGE19 = u"""

Product name

60.00

- +

description

diff --git a/scrapy/tests/test_contrib_ibl/test_htmlpage.py b/scrapy/tests/test_contrib_ibl/test_htmlpage.py index cdba56853..d85ae1afb 100644 --- a/scrapy/tests/test_contrib_ibl/test_htmlpage.py +++ b/scrapy/tests/test_contrib_ibl/test_htmlpage.py @@ -137,3 +137,19 @@ class TestParseHtml(TestCase): parsed = list(parse_html("")) self.assertEqual(parsed[0].attributes, {'src': 'http://images.play.com/banners/SAM550a.jpg', \ 'align': 'left', 'hspace': '5', '/': None}) + + def test_no_ending_body(self): + """Test case when no ending body nor html elements are present""" + parsed = [_decode_element(d) for d in PARSED7] + self._test_sample(PAGE7, parsed) + + def test_malformed(self): + """Test parsing of some malformed cases""" + parsed = [_decode_element(d) for d in PARSED8] + self._test_sample(PAGE8, parsed) + + def test_malformed2(self): + """Test case when attributes are not separated by space (still recognizable because of quotes)""" + parsed = [_decode_element(d) for d in PARSED9] + self._test_sample(PAGE9, parsed) + diff --git a/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py b/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py index 62cd6b526..f54dc9f8c 100644 --- a/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py +++ b/scrapy/tests/test_contrib_ibl/test_htmlpage_data.py @@ -246,3 +246,32 @@ PARSED7 = [ {'end': 99, 'start': 85}, ] +PAGE8 = u"""""" + +PARSED8 = [ + {'attributes' : {u'href' : u"/overview.asp?id=277"}, 'end': 31, 'start': 0, 'tag': u'a', 'tag_type': 1}, + {'attributes' : {u'src' : u"/img/5200814311.jpg", u'border' : u"0", u'title': u'Vinyl Cornice'}, 'end': 94, 'start': 31, 'tag': u'img', 'tag_type': 1}, + {'attributes' : {}, 'end': 98, 'start': 94, 'tag': u'a', 'tag_type': 2}, + {'attributes' : {}, 'end': 103, 'start': 98, 'tag': u'td', 'tag_type': 2}, + {'attributes' : {u'width': u'5'}, 'end': 120, 'start': 103, 'tag': u'table', 'tag_type': 1} +] + +PAGE9 = u"""\ +\ +\ +\ +Click here\ +\ +\ +""" + +PARSED9 = [ + {'attributes' : {}, 'end': 6, 'start': 0, 'tag': 'html', 'tag_type': 1}, + {'attributes' : {}, 'end': 12, 'start': 6, 'tag': 'body', 'tag_type': 1}, + {'attributes' : {'width': '230', 'height': '150', 'src': '/images/9589.jpg'}, 'end': 65, 'start': 12, 'tag': 'img', 'tag_type': 1}, + {'attributes' : {'href': '/product/9589'}, 'end': 89, 'start': 65, 'tag': 'a', 'tag_type': 1}, + {'end': 99, 'start': 89}, + {'attributes' : {}, 'end': 103, 'start': 99, 'tag': 'a', 'tag_type': 2}, + {'attributes' : {}, 'end': 110, 'start': 103, 'tag': 'body', 'tag_type': 2}, + {'attributes' : {}, 'end': 117, 'start': 110, 'tag': 'html', 'tag_type': 2}, +] diff --git a/scrapy/tests/test_link.py b/scrapy/tests/test_link.py new file mode 100644 index 000000000..32e0095e6 --- /dev/null +++ b/scrapy/tests/test_link.py @@ -0,0 +1,28 @@ +import unittest + +from scrapy.link import Link + +class LinkTest(unittest.TestCase): + + def test_eq_and_hash(self): + l1 = Link("http://www.example.com") + l2 = Link("http://www.example.com/other") + l3 = Link("http://www.example.com") + + self.assertEqual(l1, l1) + self.assertEqual(hash(l1), hash(l1)) + self.assertNotEqual(l1, l2) + self.assertNotEqual(hash(l1), hash(l2)) + self.assertEqual(l1, l3) + self.assertEqual(hash(l1), hash(l3)) + + l4 = Link("http://www.example.com", text="test") + l5 = Link("http://www.example.com", text="test2") + l6 = Link("http://www.example.com", text="test") + + self.assertEqual(l4, l4) + self.assertEqual(hash(l4), hash(l4)) + self.assertNotEqual(l4, l5) + self.assertNotEqual(hash(l4), hash(l5)) + self.assertEqual(l4, l6) + self.assertEqual(hash(l4), hash(l6))