From 3d4317bfe4697955de1c2450ce4cb0d12471a9a3 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 21 Oct 2019 18:32:30 +0100 Subject: [PATCH 01/34] [tox.ini] Added python 3.8 fields https://github.com/scrapy/scrapy/issues/4085 --- tox.ini | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/tox.ini b/tox.ini index ffe7360d3..14afec23f 100644 --- a/tox.ini +++ b/tox.ini @@ -92,6 +92,10 @@ deps = {[testenv:py35]deps} basepython = python3.7 deps = {[testenv:py35]deps} +[testenv:py38] +basepython = python3.8 +deps = {[testenv:py35]deps} + [testenv:pypy3] basepython = pypy3 deps = {[testenv:py35]deps} @@ -128,6 +132,13 @@ deps = reppy robotexclusionrulesparser +[testenv:py38-extra-deps] +basepython = python3.8 +deps = + {[testenv:py35]deps} + reppy + robotexclusionrulesparser + [testenv:py27-extra-deps] basepython = python2.7 deps = From 4e939ca75d14daf5bc658fdbe5a97af6f4c3c498 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 21 Oct 2019 18:33:18 +0100 Subject: [PATCH 02/34] [setup.py] Added python 3.8 fields https://github.com/scrapy/scrapy/issues/4085 --- setup.py | 1 + 1 file changed, 1 insertion(+) diff --git a/setup.py b/setup.py index 850456503..2f5fca4c9 100644 --- a/setup.py +++ b/setup.py @@ -56,6 +56,7 @@ setup( 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', + 'Programming Language :: Python :: 3.8', 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', From c12a075164e95e98a136499faf7bf8eefdce0300 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 21 Oct 2019 18:34:15 +0100 Subject: [PATCH 03/34] [.travis.yml] Added python 3.8 fields https://github.com/scrapy/scrapy/issues/4085 --- .travis.yml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.travis.yml b/.travis.yml index 0190a7f4d..2ba504972 100644 --- a/.travis.yml +++ b/.travis.yml @@ -27,6 +27,10 @@ matrix: python: 3.7 - env: TOXENV=py37-extra-deps python: 3.7 + - env: TOXENV=py38 + python: 3.8 + - env: TOXENV=py37-extra-deps + python: 3.8 - env: TOXENV=docs python: 3.6 install: From 85ac5c5c5778a116f762cb6200492aa6a2439e83 Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Mon, 21 Oct 2019 19:06:43 +0100 Subject: [PATCH 04/34] Update .travis.yml Co-Authored-By: Mikhail Korobov --- .travis.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index 2ba504972..044fa9e95 100644 --- a/.travis.yml +++ b/.travis.yml @@ -29,7 +29,7 @@ matrix: python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py37-extra-deps + - env: TOXENV=py38-extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 From 179dc916ff3e3518d8874f3fcd93948e11d5a259 Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Sun, 27 Oct 2019 16:48:20 +0000 Subject: [PATCH 05/34] Update setup.py Updating setup.py to remove python 3.8 support for now --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 2f5fca4c9..4127d3191 100644 --- a/setup.py +++ b/setup.py @@ -56,7 +56,7 @@ setup( 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', - 'Programming Language :: Python :: 3.8', + # 'Programming Language :: Python :: 3.8', not supported yet 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', From 4068797558f6d8d58f81e12930f487fc0f59f0a5 Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Sun, 27 Oct 2019 17:02:17 +0000 Subject: [PATCH 06/34] Update test_downloadermiddleware_httpcache.py Adding xfail denoting that leveldb is not supported in 3.8 --- tests/test_downloadermiddleware_httpcache.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 22946b98c..34bf5776a 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -154,6 +154,7 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) +@pytest.mark.xfail(reason='leveldb not supported in python 3.8') class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') From deacd34c8d2400b07e911dea9173b840cec7dece Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 17:39:47 +0000 Subject: [PATCH 07/34] [test_downloadermiddleware_httpcache] Attempting to add xfail for leveldb related tests https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 34bf5776a..e350da72c 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -104,6 +104,7 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): + @pytest.mark.xfail(reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -154,7 +155,6 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) -@pytest.mark.xfail(reason='leveldb not supported in python 3.8') class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') From 11942c436c78e998e3d58f0aad7cba490d74cc67 Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:02:13 +0000 Subject: [PATCH 08/34] [test_downloadermiddleware_httpcache] Trying hack to handle systemerror whjen importing leveldb https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index e350da72c..eec0feafc 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,6 +6,7 @@ import unittest import email.utils from contextlib import contextmanager import pytest +import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -157,7 +158,10 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): class LeveldbStorageTest(DefaultStorageTest): - pytest.importorskip('leveldb') + try: + pytest.importorskip('leveldb') + except SystemError: + pass storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From b3df0a84150f15ff00cdfaffc586f91b653baecd Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:28:47 +0000 Subject: [PATCH 09/34] [test_downloadermiddleware_httpcache] Adding xfails to impacted tests following hack fix https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index eec0feafc..605223088 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,7 +6,6 @@ import unittest import email.utils from contextlib import contextmanager import pytest -import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -90,6 +89,7 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) + @pytest.mark.xfail(reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -119,6 +119,7 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None + @pytest.mark.xfail(reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None @@ -161,6 +162,8 @@ class LeveldbStorageTest(DefaultStorageTest): try: pytest.importorskip('leveldb') except SystemError: + # Happens in python 3.8 + # This will cause xfail in DefaultStorageTest to trigger pass storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 70b2854590c1aa10324e6a4b50b6d74a5079c8e9 Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:51:26 +0000 Subject: [PATCH 10/34] [test_downloadermiddleware_httpcache] Making xfails more informative https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 605223088..faee37446 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,6 +6,7 @@ import unittest import email.utils from contextlib import contextmanager import pytest +import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -89,7 +90,7 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) - @pytest.mark.xfail(reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -105,7 +106,7 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): - @pytest.mark.xfail(reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -119,7 +120,7 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None - @pytest.mark.xfail(reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None From 20ea912513053aa2b2122ba5bb0387189bbc5467 Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:52:01 +0000 Subject: [PATCH 11/34] [test_downloadermiddleware_httpcache] Making xfails more informative https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index faee37446..b832fc38c 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -90,7 +90,7 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -106,7 +106,7 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -120,7 +120,7 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None From bb91f9c78c9c8c892495f6e6252cbeebbe12725b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 21 Oct 2019 12:08:35 +0200 Subject: [PATCH 12/34] Cover Scrapy 1.7.4 in the release notes --- docs/news.rst | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/docs/news.rst b/docs/news.rst index 59317f5eb..8dfe8693c 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -6,6 +6,18 @@ Release notes .. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +Scrapy 1.7.4 (2019-10-21) +------------------------- + +Revert the fix for :issue:`3804` (:issue:`3819`), which has a few undesired +side effects (:issue:`3897`, :issue:`3976`). + +As a result, when an item loader is initialized with an item, +:meth:`ItemLoader.load_item() ` once again +makes later calls to :meth:`ItemLoader.get_output_value() +` or :meth:`ItemLoader.load_item() +` return empty data. + Scrapy 1.7.3 (2019-08-01) ------------------------- From 7731814cc25c57fe31db9ba749450cd5a27eed39 Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 28 Oct 2019 06:53:53 -0300 Subject: [PATCH 13/34] ItemLoader: improve handling of initial item (#4036) --- docs/topics/loaders.rst | 10 +- scrapy/loader/__init__.py | 31 ++-- tests/test_loader.py | 321 +++++++++++++++++++++++++++++--------- 3 files changed, 273 insertions(+), 89 deletions(-) diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 1c2f1da4d..0318e37aa 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -35,6 +35,12 @@ Then, you start collecting values into the Item Loader, typically using the same item field; the Item Loader will know how to "join" those values later using a proper processing function. +.. note:: Collected data is internally stored as lists, + allowing to add several values to the same field. + If an ``item`` argument is passed when creating a loader, + each of the item's values will be stored as-is if it's already + an iterable, or wrapped with a list if it's a single value. + Here is a typical Item Loader usage in a :ref:`Spider `, using the :ref:`Product item ` declared in the :ref:`Items chapter `:: @@ -128,9 +134,9 @@ So what happens is: It's worth noticing that processors are just callable objects, which are called with the data to be parsed, and return a parsed value. So you can use any function as input or output processor. The only requirement is that they must -accept one (and only one) positional argument, which will be an iterator. +accept one (and only one) positional argument, which will be an iterable. -.. note:: Both input and output processors must receive an iterator as their +.. note:: Both input and output processors must receive an iterable as their first argument. The output of those functions can be anything. The result of input processors will be appended to an internal list (in the Loader) containing the collected values (for that field). The result of the output diff --git a/scrapy/loader/__init__.py b/scrapy/loader/__init__.py index 6665eba16..60fd6d222 100644 --- a/scrapy/loader/__init__.py +++ b/scrapy/loader/__init__.py @@ -1,19 +1,19 @@ -"""Item Loader +""" +Item Loader See documentation in docs/topics/loaders.rst - """ from collections import defaultdict + import six from scrapy.item import Item +from scrapy.loader.common import wrap_loader_context +from scrapy.loader.processors import Identity from scrapy.selector import Selector from scrapy.utils.misc import arg_to_iter, extract_regex from scrapy.utils.python import flatten -from .common import wrap_loader_context -from .processors import Identity - class ItemLoader(object): @@ -33,10 +33,9 @@ class ItemLoader(object): self.parent = parent self._local_item = context['item'] = item self._local_values = defaultdict(list) - # Preprocess values if item built from dict - # Values need to be added to item._values if added them from dict (not with add_values) + # values from initial item for field_name, value in item.items(): - self._values[field_name] = self._process_input_value(field_name, value) + self._values[field_name] += arg_to_iter(value) @property def _values(self): @@ -132,8 +131,8 @@ class ItemLoader(object): try: return proc(self._values[field_name]) except Exception as e: - raise ValueError("Error with output processor: field=%r value=%r error='%s: %s'" % \ - (field_name, self._values[field_name], type(e).__name__, str(e))) + raise ValueError("Error with output processor: field=%r value=%r error='%s: %s'" % + (field_name, self._values[field_name], type(e).__name__, str(e))) def get_collected_values(self, field_name): return self._values[field_name] @@ -141,15 +140,15 @@ class ItemLoader(object): def get_input_processor(self, field_name): proc = getattr(self, '%s_in' % field_name, None) if not proc: - proc = self._get_item_field_attr(field_name, 'input_processor', \ - self.default_input_processor) + proc = self._get_item_field_attr(field_name, 'input_processor', + self.default_input_processor) return proc def get_output_processor(self, field_name): proc = getattr(self, '%s_out' % field_name, None) if not proc: - proc = self._get_item_field_attr(field_name, 'output_processor', \ - self.default_output_processor) + proc = self._get_item_field_attr(field_name, 'output_processor', + self.default_output_processor) return proc def _process_input_value(self, field_name, value): @@ -174,8 +173,8 @@ class ItemLoader(object): def _check_selector_method(self): if self.selector is None: raise RuntimeError("To use XPath or CSS selectors, " - "%s must be instantiated with a selector " - "or a response" % self.__class__.__name__) + "%s must be instantiated with a selector " + "or a response" % self.__class__.__name__) def add_xpath(self, field_name, xpath, *processors, **kw): values = self._get_xpathvalues(xpath, **kw) diff --git a/tests/test_loader.py b/tests/test_loader.py index 2725b001a..4a4264a2a 100644 --- a/tests/test_loader.py +++ b/tests/test_loader.py @@ -1,13 +1,15 @@ -import unittest -import six from functools import partial +import unittest + +import six -from scrapy.loader import ItemLoader -from scrapy.loader.processors import Join, Identity, TakeFirst, \ - Compose, MapCompose, SelectJmes -from scrapy.item import Item, Field -from scrapy.selector import Selector from scrapy.http import HtmlResponse +from scrapy.item import Item, Field +from scrapy.loader import ItemLoader +from scrapy.loader.processors import (Compose, Identity, Join, + MapCompose, SelectJmes, TakeFirst) +from scrapy.selector import Selector + # test items class NameItem(Item): @@ -61,7 +63,7 @@ class BasicItemLoaderTest(unittest.TestCase): il.add_value('name', u'marta') item = il.load_item() assert item is i - self.assertEqual(item['summary'], u'lala') + self.assertEqual(item['summary'], [u'lala']) self.assertEqual(item['name'], [u'marta']) def test_load_item_using_custom_loader(self): @@ -419,43 +421,6 @@ class BasicItemLoaderTest(unittest.TestCase): self.assertEqual(item['url'], u'rabbit.hole') self.assertEqual(item['summary'], u'rabbithole') - def test_create_item_from_dict(self): - class TestItem(Item): - title = Field() - - class TestItemLoader(ItemLoader): - default_item_class = TestItem - - input_item = {'title': 'Test item title 1'} - il = TestItemLoader(item=input_item) - # Getting output value mustn't remove value from item - self.assertEqual(il.load_item(), { - 'title': 'Test item title 1', - }) - self.assertEqual(il.get_output_value('title'), 'Test item title 1') - self.assertEqual(il.load_item(), { - 'title': 'Test item title 1', - }) - - input_item = {'title': 'Test item title 2'} - il = TestItemLoader(item=input_item) - # Values from dict must be added to item _values - self.assertEqual(il._values.get('title'), 'Test item title 2') - - input_item = {'title': [u'Test item title 3', u'Test item 4']} - il = TestItemLoader(item=input_item) - # Same rules must work for lists - self.assertEqual(il._values.get('title'), - [u'Test item title 3', u'Test item 4']) - self.assertEqual(il.load_item(), { - 'title': [u'Test item title 3', u'Test item 4'], - }) - self.assertEqual(il.get_output_value('title'), - [u'Test item title 3', u'Test item 4']) - self.assertEqual(il.load_item(), { - 'title': [u'Test item title 3', u'Test item 4'], - }) - def test_error_input_processor(self): class TestItem(Item): name = Field() @@ -493,6 +458,220 @@ class BasicItemLoaderTest(unittest.TestCase): [u'marta', u'other'], Compose(float)) +class InitializationTestMixin(object): + + item_class = None + + def test_keep_single_value(self): + """Loaded item should contain values from the initial item""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo']}) + + def test_keep_list(self): + """Loaded item should contain values from the initial item""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']}) + + def test_add_value_singlevalue_singlevalue(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + il.add_value('name', 'bar') + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']}) + + def test_add_value_singlevalue_list(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + il.add_value('name', ['item', 'loader']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'item', 'loader']}) + + def test_add_value_list_singlevalue(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + il.add_value('name', 'qwerty') + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'qwerty']}) + + def test_add_value_list_list(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + il.add_value('name', ['item', 'loader']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'item', 'loader']}) + + def test_get_output_value_singlevalue(self): + """Getting output value must not remove value from item""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + self.assertEqual(il.get_output_value('name'), ['foo']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(loaded_item, dict({'name': ['foo']})) + + def test_get_output_value_list(self): + """Getting output value must not remove value from item""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + self.assertEqual(il.get_output_value('name'), ['foo', 'bar']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(loaded_item, dict({'name': ['foo', 'bar']})) + + def test_values_single(self): + """Values from initial item must be added to loader._values""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + self.assertEqual(il._values.get('name'), ['foo']) + + def test_values_list(self): + """Values from initial item must be added to loader._values""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + self.assertEqual(il._values.get('name'), ['foo', 'bar']) + + +class InitializationFromDictTest(InitializationTestMixin, unittest.TestCase): + item_class = dict + + +class InitializationFromItemTest(InitializationTestMixin, unittest.TestCase): + item_class = NameItem + + +class BaseNoInputReprocessingLoader(ItemLoader): + title_in = MapCompose(str.upper) + title_out = TakeFirst() + + +class NoInputReprocessingDictLoader(BaseNoInputReprocessingLoader): + default_item_class = dict + + +class NoInputReprocessingFromDictTest(unittest.TestCase): + """ + Loaders initialized from loaded items must not reprocess fields (dict instances) + """ + def test_avoid_reprocessing_with_initial_values_single(self): + il = NoInputReprocessingDictLoader(item=dict(title='foo')) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='foo')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo')) + + def test_avoid_reprocessing_with_initial_values_list(self): + il = NoInputReprocessingDictLoader(item=dict(title=['foo', 'bar'])) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='foo')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo')) + + def test_avoid_reprocessing_without_initial_values_single(self): + il = NoInputReprocessingDictLoader() + il.add_value('title', 'foo') + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='FOO')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO')) + + def test_avoid_reprocessing_without_initial_values_list(self): + il = NoInputReprocessingDictLoader() + il.add_value('title', ['foo', 'bar']) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='FOO')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO')) + + +class NoInputReprocessingItem(Item): + title = Field() + + +class NoInputReprocessingItemLoader(BaseNoInputReprocessingLoader): + default_item_class = NoInputReprocessingItem + + +class NoInputReprocessingFromItemTest(unittest.TestCase): + """ + Loaders initialized from loaded items must not reprocess fields (BaseItem instances) + """ + def test_avoid_reprocessing_with_initial_values_single(self): + il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title='foo')) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'foo'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'}) + + def test_avoid_reprocessing_with_initial_values_list(self): + il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title=['foo', 'bar'])) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'foo'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'}) + + def test_avoid_reprocessing_without_initial_values_single(self): + il = NoInputReprocessingItemLoader() + il.add_value('title', 'FOO') + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'FOO'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'}) + + def test_avoid_reprocessing_without_initial_values_list(self): + il = NoInputReprocessingItemLoader() + il.add_value('title', ['foo', 'bar']) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'FOO'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'}) + + +class TestOutputProcessorDict(unittest.TestCase): + def test_output_processor(self): + + class TempDict(dict): + def __init__(self, *args, **kwargs): + super(TempDict, self).__init__(self, *args, **kwargs) + self.setdefault('temp', 0.3) + + class TempLoader(ItemLoader): + default_item_class = TempDict + default_input_processor = Identity() + default_output_processor = Compose(TakeFirst()) + + loader = TempLoader() + item = loader.load_item() + self.assertIsInstance(item, TempDict) + self.assertEqual(dict(item), {'temp': 0.3}) + + +class TestOutputProcessorItem(unittest.TestCase): + def test_output_processor(self): + + class TempItem(Item): + temp = Field() + + def __init__(self, *args, **kwargs): + super(TempItem, self).__init__(self, *args, **kwargs) + self.setdefault('temp', 0.3) + + class TempLoader(ItemLoader): + default_item_class = TempItem + default_input_processor = Identity() + default_output_processor = Compose(TakeFirst()) + + loader = TempLoader() + item = loader.load_item() + self.assertIsInstance(item, TempItem) + self.assertEqual(dict(item), {'temp': 0.3}) + + class ProcessorsTest(unittest.TestCase): def test_take_first(self): @@ -523,7 +702,8 @@ class ProcessorsTest(unittest.TestCase): self.assertRaises(ValueError, proc, 'hello') def test_mapcompose(self): - filter_world = lambda x: None if x == 'world' else x + def filter_world(x): + return None if x == 'world' else x proc = MapCompose(filter_world, six.text_type.upper) self.assertEqual(proc([u'hello', u'world', u'this', u'is', u'scrapy']), [u'HELLO', u'THIS', u'IS', u'SCRAPY']) @@ -535,7 +715,6 @@ class ProcessorsTest(unittest.TestCase): self.assertRaises(ValueError, proc, 'hello') - class SelectortemLoaderTest(unittest.TestCase): response = HtmlResponse(url="", encoding='utf-8', body=b""" @@ -672,7 +851,7 @@ class SelectortemLoaderTest(unittest.TestCase): self.assertEqual(l.get_css(['p::text', 'div::text']), [u'paragraph', 'marta']) self.assertEqual(l.get_css(['a::attr(href)', 'img::attr(src)']), - [u'http://www.scrapy.org', u'/images/logo.png']) + [u'http://www.scrapy.org', u'/images/logo.png']) def test_replace_css_multi_fields(self): l = TestItemLoader(response=self.response) @@ -720,7 +899,7 @@ class SubselectorLoaderTest(unittest.TestCase): self.assertEqual(l.get_output_value('name'), [u'marta']) self.assertEqual(l.get_output_value('name_div'), [u'
marta
']) - self.assertEqual(l.get_output_value('name_value'), [u'marta']) + self.assertEqual(l.get_output_value('name_value'), [u'marta']) self.assertEqual(l.get_output_value('name'), nl.get_output_value('name')) self.assertEqual(l.get_output_value('name_div'), nl.get_output_value('name_div')) @@ -735,7 +914,7 @@ class SubselectorLoaderTest(unittest.TestCase): self.assertEqual(l.get_output_value('name'), [u'marta']) self.assertEqual(l.get_output_value('name_div'), [u'
marta
']) - self.assertEqual(l.get_output_value('name_value'), [u'marta']) + self.assertEqual(l.get_output_value('name_value'), [u'marta']) self.assertEqual(l.get_output_value('name'), nl.get_output_value('name')) self.assertEqual(l.get_output_value('name_div'), nl.get_output_value('name_div')) @@ -791,28 +970,28 @@ class SubselectorLoaderTest(unittest.TestCase): class SelectJmesTestCase(unittest.TestCase): - test_list_equals = { - 'simple': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), - 'invalid': ('foo.bar.baz', {"foo": {"bar": "baz"}}, None), - 'top_level': ('foo', {"foo": {"bar": "baz"}}, {"bar": "baz"}), - 'double_vs_single_quote_string': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), - 'dict': ( - 'foo.bar[*].name', - {"foo": {"bar": [{"name": "one"}, {"name": "two"}]}}, - ['one', 'two'] - ), - 'list': ('[1]', [1, 2], 2) - } + test_list_equals = { + 'simple': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), + 'invalid': ('foo.bar.baz', {"foo": {"bar": "baz"}}, None), + 'top_level': ('foo', {"foo": {"bar": "baz"}}, {"bar": "baz"}), + 'double_vs_single_quote_string': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), + 'dict': ( + 'foo.bar[*].name', + {"foo": {"bar": [{"name": "one"}, {"name": "two"}]}}, + ['one', 'two'] + ), + 'list': ('[1]', [1, 2], 2) + } - def test_output(self): - for l in self.test_list_equals: - expr, test_list, expected = self.test_list_equals[l] - test = SelectJmes(expr)(test_list) - self.assertEqual( - test, - expected, - msg='test "{}" got {} expected {}'.format(l, test, expected) - ) + def test_output(self): + for l in self.test_list_equals: + expr, test_list, expected = self.test_list_equals[l] + test = SelectJmes(expr)(test_list) + self.assertEqual( + test, + expected, + msg='test "{}" got {} expected {}'.format(l, test, expected) + ) if __name__ == "__main__": From b5a00262ec48534b89750037060318326b4e349c Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Mon, 28 Oct 2019 09:59:33 +0000 Subject: [PATCH 14/34] Update .travis.yml Reverting change to 3.8 extra dependency environment. --- .travis.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index 044fa9e95..2ba504972 100644 --- a/.travis.yml +++ b/.travis.yml @@ -29,7 +29,7 @@ matrix: python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py38-extra-deps + - env: TOXENV=py37-extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 From 3d0df419c4eeb0ccf7934c8f06a38707eae8d722 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 28 Oct 2019 11:24:47 +0100 Subject: [PATCH 15/34] Mark the LevelDB storage backend as deprecated --- docs/topics/downloader-middleware.rst | 22 ---------------------- scrapy/extensions/httpcache.py | 17 ++++++++++++----- 2 files changed, 12 insertions(+), 27 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 2892b9b79..8048e1c86 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -348,7 +348,6 @@ HttpCacheMiddleware * :ref:`httpcache-storage-fs` * :ref:`httpcache-storage-dbm` - * :ref:`httpcache-storage-leveldb` You can change the HTTP cache storage backend with the :setting:`HTTPCACHE_STORAGE` setting. Or you can also :ref:`implement your own storage backend. ` @@ -478,27 +477,6 @@ DBM storage backend By default, it uses the anydbm_ module, but you can change it with the :setting:`HTTPCACHE_DBM_MODULE` setting. -.. _httpcache-storage-leveldb: - -LevelDB storage backend -~~~~~~~~~~~~~~~~~~~~~~~ - -.. class:: LeveldbCacheStorage - - .. versionadded:: 0.23 - - A LevelDB_ storage backend is also available for the HTTP cache middleware. - - This backend is not recommended for development because only one process - can access LevelDB databases at the same time, so you can't run a crawl and - open the scrapy shell in parallel for the same spider. - - In order to use this storage backend, install the `LevelDB python - bindings`_ (e.g. ``pip install leveldb``). - - .. _LevelDB: https://github.com/google/leveldb - .. _leveldb python bindings: https://pypi.python.org/pypi/leveldb - .. _httpcache-storage-custom: Writing your own storage backend diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index c6094643d..7c650a91e 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -1,19 +1,24 @@ from __future__ import print_function -import os + import gzip import logging -from six.moves import cPickle as pickle +import os +from email.utils import mktime_tz, parsedate_tz from importlib import import_module from time import time +from warnings import warn from weakref import WeakKeyDictionary -from email.utils import mktime_tz, parsedate_tz + +from six.moves import cPickle as pickle from w3lib.http import headers_raw_to_dict, headers_dict_to_raw + +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Headers, Response from scrapy.responsetypes import responsetypes -from scrapy.utils.request import request_fingerprint -from scrapy.utils.project import data_path from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.project import data_path from scrapy.utils.python import to_bytes, to_unicode, garbage_collect +from scrapy.utils.request import request_fingerprint logger = logging.getLogger(__name__) @@ -345,6 +350,8 @@ class FilesystemCacheStorage(object): class LeveldbCacheStorage(object): def __init__(self, settings): + warn("The LevelDB storage backend is deprecated.", + ScrapyDeprecationWarning, stacklevel=2) import leveldb self._leveldb = leveldb self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True) From 16bb3ac20dae8b7c5fbccf4ab85b3a0393e7c55d Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 11:24:09 +0000 Subject: [PATCH 16/34] [test_downloadermiddleware_httpcache] Using skipif approach https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index b832fc38c..ba5027307 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -90,7 +90,6 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -106,7 +105,6 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -120,7 +118,6 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None @@ -164,8 +161,10 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - # This will cause xfail in DefaultStorageTest to trigger - pass + pytest.mark.skipif( + sys.version_info >= (3, 8), + reason='leveldb not supported in python 3.8', + ) storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 9b47dc6a703310d13c9470e50d4b14f81ee893c6 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 11:24:52 +0000 Subject: [PATCH 17/34] [travis, setup] Adding official python 3.8 support https://github.com/scrapy/scrapy/issues/4085 --- .travis.yml | 4 +--- setup.py | 2 +- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/.travis.yml b/.travis.yml index 2ba504972..4c2498053 100644 --- a/.travis.yml +++ b/.travis.yml @@ -25,11 +25,9 @@ matrix: python: 3.6 - env: TOXENV=py37 python: 3.7 - - env: TOXENV=py37-extra-deps - python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py37-extra-deps + - env: TOXENV=py38-extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 diff --git a/setup.py b/setup.py index 4127d3191..2f5fca4c9 100644 --- a/setup.py +++ b/setup.py @@ -56,7 +56,7 @@ setup( 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', - # 'Programming Language :: Python :: 3.8', not supported yet + 'Programming Language :: Python :: 3.8', 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', From 4432136ffff4d8af42f7a485c17ab7fbbb228078 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:22:21 +0000 Subject: [PATCH 18/34] [test_downloadermiddleware_httpcache] Fixing pytest skip behaviour https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index ba5027307..32085b095 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,7 +6,6 @@ import unittest import email.utils from contextlib import contextmanager import pytest -import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -161,10 +160,7 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - pytest.mark.skipif( - sys.version_info >= (3, 8), - reason='leveldb not supported in python 3.8', - ) + pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From c51fb959e2985faf6f21fe7f03d2fb8160de064f Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:36:54 +0000 Subject: [PATCH 19/34] [test_downloadermiddleware_httpcache] Fixing pytest skip behaviour https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 32085b095..047526569 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,6 +6,7 @@ import unittest import email.utils from contextlib import contextmanager import pytest +import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -160,7 +161,7 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") + pytestmark = pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 74909030a55b59e3b858fc736b5b1f685d9596a6 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:52:36 +0000 Subject: [PATCH 20/34] [tox.ini] Removing obsolete py37 extra deps enviornment https://github.com/scrapy/scrapy/issues/4085 --- tox.ini | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tox.ini b/tox.ini index 14afec23f..fe925951b 100644 --- a/tox.ini +++ b/tox.ini @@ -125,13 +125,6 @@ deps = {[docs]deps} commands = sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck -[testenv:py37-extra-deps] -basepython = python3.7 -deps = - {[testenv:py35]deps} - reppy - robotexclusionrulesparser - [testenv:py38-extra-deps] basepython = python3.8 deps = From b73d217de5647a68c7b8dfda747cd3d0685c226d Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:55:54 +0000 Subject: [PATCH 21/34] [test_downloadermiddleware_httpcache.py] Fixing pytest mark behaviour https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 047526569..f5917d0f0 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -161,7 +161,7 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - pytestmark = pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") + pytestmark = pytest.mark.skip("'SystemError: bad call flags' occurs on Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 93e3dc1b826e44d1a5a24fbb39c090ce426aa862 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 16:12:03 +0000 Subject: [PATCH 22/34] [test_downloadermiddleware_httpcache.py] Cleaning text https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index f5917d0f0..972d400a4 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -155,13 +155,13 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) + class LeveldbStorageTest(DefaultStorageTest): try: pytest.importorskip('leveldb') except SystemError: - # Happens in python 3.8 - pytestmark = pytest.mark.skip("'SystemError: bad call flags' occurs on Python 3.8") + pytestmark = pytest.mark.skip("Test module skipped - 'SystemError: bad call flags' occurs when >= Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 94f060fcc84853f28f3f91b6dde1d61c8e19251e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 29 Oct 2019 12:53:46 +0100 Subject: [PATCH 23/34] Cover Scrapy 1.8.0 in the release notes (#3952) --- docs/news.rst | 226 +++++++++++++++++++++++++++++++++++++++- docs/topics/logging.rst | 5 +- scrapy/logformatter.py | 2 +- 3 files changed, 229 insertions(+), 4 deletions(-) diff --git a/docs/news.rst b/docs/news.rst index 8dfe8693c..669844045 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -6,6 +6,209 @@ Release notes .. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +.. _release-1.8.0: + +Scrapy 1.8.0 (2019-10-28) +------------------------- + +Highlights: + +* Dropped Python 3.4 support and updated minimum requirements; made Python 3.8 + support official +* New :meth:`Request.from_curl ` class method +* New :setting:`ROBOTSTXT_PARSER` and :setting:`ROBOTSTXT_USER_AGENT` settings +* New :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` and + :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` settings + +Backward-incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Python 3.4 is no longer supported, and some of the minimum requirements of + Scrapy have also changed: + + * cssselect_ 0.9.1 + * cryptography_ 2.0 + * lxml_ 3.5.0 + * pyOpenSSL_ 16.2.0 + * queuelib_ 1.4.2 + * service_identity_ 16.0.0 + * six_ 1.10.0 + * Twisted_ 17.9.0 (16.0.0 with Python 2) + * zope.interface_ 4.1.3 + + (:issue:`3892`) + +* ``JSONRequest`` is now called :class:`~scrapy.http.JsonRequest` for + consistency with similar classes (:issue:`3929`, :issue:`3982`) + +* If you are using a custom context factory + (:setting:`DOWNLOADER_CLIENTCONTEXTFACTORY`), its ``__init__`` method must + accept two new parameters: ``tls_verbose_logging`` and ``tls_ciphers`` + (:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`) + +* :class:`~scrapy.loader.ItemLoader` now turns the values of its input item + into lists:: + + >>> item = MyItem() + >>> item['field'] = 'value1' + >>> loader = ItemLoader(item=item) + >>> item['field'] + ['value1'] + + This is needed to allow adding values to existing fields + (``loader.add_value('field', 'value2')``). + + (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, :issue:`3998`, + :issue:`4036`) + +See also :ref:`1.8-deprecation-removals` below. + + +New features +~~~~~~~~~~~~ + +* A new :meth:`Request.from_curl ` class + method allows :ref:`creating a request from a cURL command + ` (:issue:`2985`, :issue:`3862`) + +* A new :setting:`ROBOTSTXT_PARSER` setting allows choosing which robots.txt_ + parser to use. It includes built-in support for + :ref:`RobotFileParser `, + :ref:`Protego ` (default), :ref:`Reppy `, and + :ref:`Robotexclusionrulesparser `, and allows you to + :ref:`implement support for additional parsers + ` (:issue:`754`, :issue:`2669`, + :issue:`3796`, :issue:`3935`, :issue:`3969`, :issue:`4006`) + +* A new :setting:`ROBOTSTXT_USER_AGENT` setting allows defining a separate + user agent string to use for robots.txt_ parsing (:issue:`3931`, + :issue:`3966`) + +* :class:`~scrapy.spiders.Rule` no longer requires a :class:`LinkExtractor + ` parameter + (:issue:`781`, :issue:`4016`) + +* Use the new :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` setting to customize + the TLS/SSL ciphers used by the default HTTP/1.1 downloader (:issue:`3392`, + :issue:`3442`) + +* Set the new :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` setting to + ``True`` to enable debug-level messages about TLS connection parameters + after establishing HTTPS connections (:issue:`2111`, :issue:`3450`) + +* Callbacks that receive keyword arguments + (see :attr:`Request.cb_kwargs `) can now be + tested using the new :class:`@cb_kwargs + ` + :ref:`spider contract ` (:issue:`3985`, :issue:`3988`) + +* When a :class:`@scrapes ` spider + contract fails, all missing fields are now reported (:issue:`766`, + :issue:`3939`) + +* :ref:`Custom log formats ` can now drop messages by + having the corresponding methods of the configured :setting:`LOG_FORMATTER` + return ``None`` (:issue:`3984`, :issue:`3987`) + +* A much improved completion definition is now available for Zsh_ + (:issue:`4069`) + + +Bug fixes +~~~~~~~~~ + +* :meth:`ItemLoader.load_item() ` no + longer makes later calls to :meth:`ItemLoader.get_output_value() + ` or + :meth:`ItemLoader.load_item() ` return + empty data (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, + :issue:`3998`, :issue:`4036`) + +* Fixed :class:`~scrapy.statscollectors.DummyStatsCollector` raising a + :exc:`TypeError` exception (:issue:`4007`, :issue:`4052`) + +* :meth:`FilesPipeline.file_path + ` and + :meth:`ImagesPipeline.file_path + ` no longer choose + file extensions that are not `registered with IANA`_ (:issue:`1287`, + :issue:`3953`, :issue:`3954`) + +* When using botocore_ to persist files in S3, all botocore-supported headers + are properly mapped now (:issue:`3904`, :issue:`3905`) + +* FTP passwords in :setting:`FEED_URI` containing percent-escaped characters + are now properly decoded (:issue:`3941`) + +* A memory-handling and error-handling issue in + :func:`scrapy.utils.ssl.get_temp_key_info` has been fixed (:issue:`3920`) + + +Documentation +~~~~~~~~~~~~~ + +* The documentation now covers how to define and configure a :ref:`custom log + format ` (:issue:`3616`, :issue:`3660`) + +* API documentation added for :class:`~scrapy.exporters.MarshalItemExporter` + and :class:`~scrapy.exporters.PythonItemExporter` (:issue:`3973`) + +* API documentation added for :class:`~scrapy.item.BaseItem` and + :class:`~scrapy.item.ItemMeta` (:issue:`3999`) + +* Minor documentation fixes (:issue:`2998`, :issue:`3398`, :issue:`3597`, + :issue:`3894`, :issue:`3934`, :issue:`3978`, :issue:`3993`, :issue:`4022`, + :issue:`4028`, :issue:`4033`, :issue:`4046`, :issue:`4050`, :issue:`4055`, + :issue:`4056`, :issue:`4061`, :issue:`4072`, :issue:`4071`, :issue:`4079`, + :issue:`4081`, :issue:`4089`, :issue:`4093`) + + +.. _1.8-deprecation-removals: + +Deprecation removals +~~~~~~~~~~~~~~~~~~~~ + +* ``scrapy.xlib`` has been removed (:issue:`4015`) + + +Deprecations +~~~~~~~~~~~~ + +* The LevelDB_ storage backend + (``scrapy.extensions.httpcache.LeveldbCacheStorage``) of + :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` is + deprecated (:issue:`4085`, :issue:`4092`) + +* Use of the undocumented ``SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE`` environment + variable is deprecated (:issue:`3910`) + +* ``scrapy.item.DictItem`` is deprecated, use :class:`~scrapy.item.Item` + instead (:issue:`3999`) + + +Other changes +~~~~~~~~~~~~~ + +* Minimum versions of optional Scrapy requirements that are covered by + continuous integration tests have been updated: + + * botocore_ 1.3.23 + * Pillow_ 3.4.2 + + Lower versions of these optional requirements may work, but it is not + guaranteed (:issue:`3892`) + +* GitHub templates for bug reports and feature requests (:issue:`3126`, + :issue:`3471`, :issue:`3749`, :issue:`3754`) + +* Continuous integration fixes (:issue:`3923`) + +* Code cleanup (:issue:`3391`, :issue:`3907`, :issue:`3946`, :issue:`3950`, + :issue:`4023`, :issue:`4031`) + + +.. _release-1.7.4: + Scrapy 1.7.4 (2019-10-21) ------------------------- @@ -18,22 +221,31 @@ makes later calls to :meth:`ItemLoader.get_output_value() ` or :meth:`ItemLoader.load_item() ` return empty data. + +.. _release-1.7.3: + Scrapy 1.7.3 (2019-08-01) ------------------------- Enforce lxml 4.3.5 or lower for Python 3.4 (:issue:`3912`, :issue:`3918`). + +.. _release-1.7.2: + Scrapy 1.7.2 (2019-07-23) ------------------------- Fix Python 2 support (:issue:`3889`, :issue:`3893`, :issue:`3896`). +.. _release-1.7.1: + Scrapy 1.7.1 (2019-07-18) ------------------------- Re-packaging of Scrapy 1.7.0, which was missing some changes in PyPI. + .. _release-1.7.0: Scrapy 1.7.0 (2019-07-18) @@ -568,7 +780,7 @@ Scrapy 1.5.2 (2019-01-22) See :ref:`telnet console ` documentation for more info -* Backport CI build failure under GCE environemnt due to boto import error. +* Backport CI build failure under GCE environment due to boto import error. .. _release-1.5.1: @@ -2830,23 +3042,35 @@ First release of Scrapy. .. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1 +.. _botocore: https://github.com/boto/botocore .. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding .. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/ .. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request +.. _cryptography: https://cryptography.io/en/latest/ .. _cssselect: https://github.com/scrapy/cssselect/ .. _docstrings: https://docs.python.org/glossary.html#term-docstring .. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt +.. _LevelDB: https://github.com/google/leveldb .. _lxml: http://lxml.de/ .. _marshal: https://docs.python.org/2/library/marshal.html .. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator .. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator .. _parsel.csstranslator.XPathExpr: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.XPathExpr .. _PEP 257: https://www.python.org/dev/peps/pep-0257/ +.. _Pillow: https://python-pillow.org/ +.. _pyOpenSSL: https://www.pyopenssl.org/en/stable/ .. _queuelib: https://github.com/scrapy/queuelib +.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml .. _resource: https://docs.python.org/2/library/resource.html +.. _robots.txt: http://www.robotstxt.org/ .. _scrapely: https://github.com/scrapy/scrapely +.. _service_identity: https://service-identity.readthedocs.io/en/stable/ +.. _six: https://six.readthedocs.io/ .. _tox: https://pypi.python.org/pypi/tox +.. _Twisted: https://twistedmatrix.com/trac/ .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _w3lib: https://github.com/scrapy/w3lib .. _w3lib.encoding: https://github.com/scrapy/w3lib/blob/master/w3lib/encoding.py .. _What is cacheable: https://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9.1 +.. _zope.interface: https://zopeinterface.readthedocs.io/en/latest/ +.. _Zsh: https://www.zsh.org/ diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index 87ea43c7d..2db0ffddd 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -198,8 +198,9 @@ to override some of the Scrapy settings regarding logging. Custom Log Formats ------------------ -A custom log format can be set for different actions by extending :class:`~scrapy.logformatter.LogFormatter` class -and making :setting:`LOG_FORMATTER` point to your new class. +A custom log format can be set for different actions by extending +:class:`~scrapy.logformatter.LogFormatter` class and making +:setting:`LOG_FORMATTER` point to your new class. .. autoclass:: scrapy.logformatter.LogFormatter :members: diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index f15940ed1..3c61ed7e0 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -29,7 +29,7 @@ class LogFormatter(object): * ``args`` should be a tuple or dict with the formatting placeholders for ``msg``. The final log message is computed as ``msg % args``. - Users can define their own ``LogFormatter`` class if they want to customise how + Users can define their own ``LogFormatter`` class if they want to customize how each action is logged or if they want to omit it entirely. In order to omit logging an action the method must return ``None``. From be2e910dd06ba4904e7b10eb5a7e3251e8dab099 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 29 Oct 2019 12:57:02 +0100 Subject: [PATCH 24/34] =?UTF-8?q?Bump=20version:=201.7.0=20=E2=86=92=201.8?= =?UTF-8?q?.0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.cfg | 2 +- scrapy/VERSION | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/.bumpversion.cfg b/.bumpversion.cfg index 70affe63f..c9f1abea5 100644 --- a/.bumpversion.cfg +++ b/.bumpversion.cfg @@ -1,5 +1,5 @@ [bumpversion] -current_version = 1.7.0 +current_version = 1.8.0 commit = True tag = True tag_name = {new_version} diff --git a/scrapy/VERSION b/scrapy/VERSION index bd8bf882d..27f9cd322 100644 --- a/scrapy/VERSION +++ b/scrapy/VERSION @@ -1 +1 @@ -1.7.0 +1.8.0 From 66cbceeb0a9104fc0fa238898e38d0d9ce9cbcf6 Mon Sep 17 00:00:00 2001 From: Amardeep Bhowmick Date: Wed, 30 Oct 2019 13:39:12 +0530 Subject: [PATCH 25/34] Fix redirection error when the Location header value starts with 3 slashes (#4042) --- scrapy/downloadermiddlewares/redirect.py | 7 +++++-- tests/test_downloadermiddleware_redirect.py | 16 ++++++++++++++++ 2 files changed, 21 insertions(+), 2 deletions(-) diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 49468a2e4..b73f864dd 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -1,5 +1,5 @@ import logging -from six.moves.urllib.parse import urljoin +from six.moves.urllib.parse import urljoin, urlparse from w3lib.url import safe_url_string @@ -70,7 +70,10 @@ class RedirectMiddleware(BaseRedirectMiddleware): if 'Location' not in response.headers or response.status not in allowed_status: return response - location = safe_url_string(response.headers['location']) + location = safe_url_string(response.headers['Location']) + if response.headers['Location'].startswith(b'//'): + request_scheme = urlparse(request.url).scheme + location = request_scheme + '://' + location.lstrip('/') redirected_url = urljoin(request.url, location) diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index 0e841489d..e7faf14a7 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -106,6 +106,22 @@ class RedirectMiddlewareTest(unittest.TestCase): del rsp.headers['Location'] assert self.mw.process_response(req, rsp, self.spider) is rsp + def test_redirect_302_relative(self): + url = 'http://www.example.com/302' + url2 = '///i8n.example2.com/302' + url3 = 'http://i8n.example2.com/302' + req = Request(url, method='HEAD') + rsp = Response(url, headers={'Location': url2}, status=302) + + req2 = self.mw.process_response(req, rsp, self.spider) + assert isinstance(req2, Request) + self.assertEqual(req2.url, url3) + self.assertEqual(req2.method, 'HEAD') + + # response without Location header but with status code is 3XX should be ignored + del rsp.headers['Location'] + assert self.mw.process_response(req, rsp, self.spider) is rsp + def test_max_redirect_times(self): self.mw.max_redirect_times = 1 From 6d6da78eda3cc0bba1bfdf70194fdf655fac8aeb Mon Sep 17 00:00:00 2001 From: Benjamin Ooghe-Tabanou Date: Wed, 30 Oct 2019 09:13:36 +0100 Subject: [PATCH 26/34] Add a keep_fragments parameter to the request_fingerprint function (#4104) --- scrapy/utils/request.py | 16 +++++++++++----- tests/test_utils_request.py | 9 ++++++++- 2 files changed, 19 insertions(+), 6 deletions(-) diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index 9c143b83a..fb5af66a2 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -16,7 +16,7 @@ from scrapy.utils.httpobj import urlparse_cached _fingerprint_cache = weakref.WeakKeyDictionary() -def request_fingerprint(request, include_headers=None): +def request_fingerprint(request, include_headers=None, keep_fragments=False): """ Return the request fingerprint. @@ -42,15 +42,21 @@ def request_fingerprint(request, include_headers=None): the fingeprint. If you want to include specific headers use the include_headers argument, which is a list of Request headers to include. + Also, servers usually ignore fragments in urls when handling requests, + so they are also ignored by default when calculating the fingerprint. + If you want to include them, set the keep_fragments argument to True + (for instance when handling requests with a headless browser). + """ if include_headers: include_headers = tuple(to_bytes(h.lower()) for h in sorted(include_headers)) cache = _fingerprint_cache.setdefault(request, {}) - if include_headers not in cache: + cache_key = (include_headers, keep_fragments) + if cache_key not in cache: fp = hashlib.sha1() fp.update(to_bytes(request.method)) - fp.update(to_bytes(canonicalize_url(request.url))) + fp.update(to_bytes(canonicalize_url(request.url, keep_fragments=keep_fragments))) fp.update(request.body or b'') if include_headers: for hdr in include_headers: @@ -58,8 +64,8 @@ def request_fingerprint(request, include_headers=None): fp.update(hdr) for v in request.headers.getlist(hdr): fp.update(v) - cache[include_headers] = fp.hexdigest() - return cache[include_headers] + cache[cache_key] = fp.hexdigest() + return cache[cache_key] def request_authenticate(request, username, password): diff --git a/tests/test_utils_request.py b/tests/test_utils_request.py index e8a4eb3ea..625a32048 100644 --- a/tests/test_utils_request.py +++ b/tests/test_utils_request.py @@ -17,7 +17,7 @@ class UtilsRequestTest(unittest.TestCase): self.assertNotEqual(request_fingerprint(r1), request_fingerprint(r2)) # make sure caching is working - self.assertEqual(request_fingerprint(r1), _fingerprint_cache[r1][None]) + self.assertEqual(request_fingerprint(r1), _fingerprint_cache[r1][(None, False)]) r1 = Request("http://www.example.com/members/offers.html") r2 = Request("http://www.example.com/members/offers.html") @@ -42,6 +42,13 @@ class UtilsRequestTest(unittest.TestCase): self.assertEqual(request_fingerprint(r3, include_headers=['accept-language', 'sessionid']), request_fingerprint(r3, include_headers=['SESSIONID', 'Accept-Language'])) + r1 = Request("http://www.example.com/test.html") + r2 = Request("http://www.example.com/test.html#fragment") + self.assertEqual(request_fingerprint(r1), request_fingerprint(r2)) + self.assertEqual(request_fingerprint(r1), request_fingerprint(r1, keep_fragments=True)) + self.assertNotEqual(request_fingerprint(r2), request_fingerprint(r2, keep_fragments=True)) + self.assertNotEqual(request_fingerprint(r1), request_fingerprint(r2, keep_fragments=True)) + r1 = Request("http://www.example.com") r2 = Request("http://www.example.com", method='POST') r3 = Request("http://www.example.com", method='POST', body=b'request body') From 229e722a03aced0fb62ec8c631b036c3e13e2188 Mon Sep 17 00:00:00 2001 From: Andrey Rahmatullin Date: Thu, 31 Oct 2019 14:46:02 +0500 Subject: [PATCH 27/34] Initial Python 2 removal (#4091) --- .travis.yml | 10 +----- README.rst | 2 +- docs/faq.rst | 4 +-- docs/intro/install.rst | 16 +++------- docs/topics/feed-exports.rst | 7 ++--- docs/topics/leaks.rst | 2 +- docs/topics/media-pipeline.rst | 2 +- docs/topics/request-response.rst | 4 +-- docs/topics/settings.rst | 5 +-- docs/topics/spiders.rst | 2 -- requirements-py2.txt | 18 ----------- scrapy/__init__.py | 4 +-- setup.py | 7 ++--- tests/requirements-py2.txt | 15 --------- tox.ini | 53 ++------------------------------ 15 files changed, 24 insertions(+), 127 deletions(-) delete mode 100644 requirements-py2.txt delete mode 100644 tests/requirements-py2.txt diff --git a/.travis.yml b/.travis.yml index 4c2498053..2352ef124 100644 --- a/.travis.yml +++ b/.travis.yml @@ -7,14 +7,6 @@ branches: - /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/ matrix: include: - - env: TOXENV=py27 - python: 2.7 - - env: TOXENV=py27-pinned - python: 2.7 - - env: TOXENV=py27-extra-deps - python: 2.7 - - env: TOXENV=pypy - python: 2.7 - env: TOXENV=pypy3 python: 3.5 - env: TOXENV=py35 @@ -70,4 +62,4 @@ deploy: on: tags: true repo: scrapy/scrapy - condition: "$TOXENV == py27 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" + condition: "$TOXENV == py37 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" diff --git a/README.rst b/README.rst index bd82bff06..fb4ca8e4f 100644 --- a/README.rst +++ b/README.rst @@ -40,7 +40,7 @@ https://scrapy.org Requirements ============ -* Python 2.7 or Python 3.5+ +* Python 3.5+ * Works on Linux, Windows, Mac OSX, BSD Install diff --git a/docs/faq.rst b/docs/faq.rst index 9733471bf..080d81981 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -69,11 +69,11 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars What Python versions does Scrapy support? ----------------------------------------- -Scrapy is supported under Python 2.7 and Python 3.5+ +Scrapy is supported under Python 3.5+ under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). -Python 2.6 support was dropped starting at Scrapy 0.20. Python 3 support was added in Scrapy 1.1. PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5. +Python 2 support was dropped in Scrapy 2.0. .. note:: For Python 3 support on Windows, it is recommended to use diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 51b41b4d7..e924b5303 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -7,7 +7,7 @@ Installation guide Installing Scrapy ================= -Scrapy runs on Python 2.7 and Python 3.5 or above +Scrapy runs on Python 3.5 or above under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). If you're using `Anaconda`_ or `Miniconda`_, you can install the package from @@ -102,10 +102,8 @@ just like any other Python package. (See :ref:`platform-specific guides ` below for non-Python dependencies that you may need to install beforehand). -Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. - -* If you want to install scrapy with Python 3, install scrapy within a Python 3 virtualenv. -* And if you want to install scrapy with Python 2, install scrapy within a Python 2 virtualenv. +Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy +only supports Python 3, make sure you created a Python 3 virtualenv. .. _virtualenv: https://virtualenv.pypa.io .. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/ @@ -149,16 +147,12 @@ typically too old and slow to catch up with latest Scrapy. To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install these dependencies:: - sudo apt-get install python-dev python-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev + sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev -- ``python-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev`` +- ``python3-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev`` are required for ``lxml`` - ``libssl-dev`` and ``libffi-dev`` are required for ``cryptography`` -If you want to install scrapy on Python 3, you’ll also need Python 3 development headers:: - - sudo apt-get install python3 python3-dev - Inside a :ref:`virtualenv `, you can install Scrapy with ``pip`` after that:: diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index af541db78..7481b1a99 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -99,12 +99,12 @@ The storages backends supported out of the box are: * :ref:`topics-feed-storage-fs` * :ref:`topics-feed-storage-ftp` - * :ref:`topics-feed-storage-s3` (requires botocore_ or boto_) + * :ref:`topics-feed-storage-s3` (requires botocore_) * :ref:`topics-feed-storage-stdout` Some storage backends may be unavailable if the required external libraries are not available. For example, the S3 backend is only available if the botocore_ -or boto_ library is installed (Scrapy supports boto_ only on Python 2). +library is installed. .. _topics-feed-uri-params: @@ -182,7 +182,7 @@ The feeds are stored on `Amazon S3`_. * ``s3://mybucket/path/to/export.csv`` * ``s3://aws_key:aws_secret@mybucket/path/to/export.csv`` - * Required external libraries: `botocore`_ (Python 2 and Python 3) or `boto`_ (Python 2 only) + * Required external libraries: `botocore`_ The AWS credentials can be passed as user/password in the URI, or they can be passed through the following settings: @@ -399,6 +399,5 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter .. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier .. _Amazon S3: https://aws.amazon.com/s3/ -.. _boto: https://github.com/boto/boto .. _botocore: https://github.com/boto/botocore .. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 8278e9849..793636f59 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -260,7 +260,7 @@ knowledge about Python internals. For more info about Guppy, refer to the Debugging memory leaks with muppy ================================= -If you're using Python 3, you can use muppy from `Pympler`_. +You can use muppy from `Pympler`_. .. _Pympler: https://pypi.org/project/Pympler/ diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 0ce431ff5..431cc6027 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -171,7 +171,7 @@ policy:: For more information, see `canned ACLs`_ in the Amazon S3 Developer Guide. -Because Scrapy uses ``boto`` / ``botocore`` internally you can also use other S3-like storages. Storages like +Because Scrapy uses ``botocore`` internally you can also use other S3-like storages. Storages like self-hosted `Minio`_ or `s3.scality`_. All you need to do is set endpoint option in you Scrapy settings:: AWS_ENDPOINT_URL = 'http://minio.example.com:9000' diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 727c67482..5a76e189e 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -596,8 +596,8 @@ Response objects (for single valued headers) or lists (for multi-valued headers). :type headers: dict - :param body: the response body. To access the decoded text as str (unicode - in Python 2) you can use ``response.text`` from an encoding-aware + :param body: the response body. To access the decoded text as str you can use + ``response.text`` from an encoding-aware :ref:`Response subclass `, such as :class:`TextResponse`. :type body: bytes diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 56375664f..a1d15a760 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -188,7 +188,6 @@ AWS_ENDPOINT_URL Default: ``None`` Endpoint URL used for S3-like storage, for example Minio or s3.scality. -Only supported with ``botocore`` library. .. setting:: AWS_USE_SSL @@ -199,7 +198,6 @@ Default: ``None`` Use this option if you want to disable SSL connection for communication with S3 or S3-like storage. By default SSL will be used. -Only supported with ``botocore`` library. .. setting:: AWS_VERIFY @@ -209,7 +207,7 @@ AWS_VERIFY Default: ``None`` Verify SSL connection between Scrapy and S3 or S3-like storage. By default -SSL verification will occur. Only supported with ``botocore`` library. +SSL verification will occur. .. setting:: AWS_REGION_NAME @@ -219,7 +217,6 @@ AWS_REGION_NAME Default: ``None`` The name of the region associated with the AWS client. -Only supported with ``botocore`` library. .. setting:: BOT_NAME diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index d60c93be6..d65a43afd 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -72,8 +72,6 @@ scrapy.Spider spider that crawls ``mywebsite.com`` would often be called ``mywebsite``. - .. note:: In Python 2 this must be ASCII only. - .. attribute:: allowed_domains An optional list of strings containing domains that this spider is diff --git a/requirements-py2.txt b/requirements-py2.txt deleted file mode 100644 index dde8d1c9c..000000000 --- a/requirements-py2.txt +++ /dev/null @@ -1,18 +0,0 @@ -parsel>=1.5.0 -PyDispatcher>=2.0.5 -w3lib>=1.17.0 -protego>=0.1.15 - -pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'" -queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'" -cryptography>=2.0 # Earlier versions would fail to install - -# Reference versions taken from -# https://packages.ubuntu.com/xenial/python/ -# https://packages.ubuntu.com/xenial/zope/ -cssselect>=0.9.1 -lxml>=3.5.0 -service_identity>=16.0.0 -six>=1.10.0 -Twisted>=16.0.0 -zope.interface>=4.1.3 diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 03ec6c667..230e5cee3 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -14,8 +14,8 @@ del pkgutil # Check minimum required Python version import sys -if sys.version_info < (2, 7): - print("Scrapy %s requires Python 2.7" % __version__) +if sys.version_info < (3, 5): + print("Scrapy %s requires Python 3.5" % __version__) sys.exit(1) # Ignore noisy twisted deprecation warnings diff --git a/setup.py b/setup.py index 2f5fca4c9..8f5f14f0d 100644 --- a/setup.py +++ b/setup.py @@ -50,8 +50,6 @@ setup( 'License :: OSI Approved :: BSD License', 'Operating System :: OS Independent', 'Programming Language :: Python', - 'Programming Language :: Python :: 2', - 'Programming Language :: Python :: 2.7', 'Programming Language :: Python :: 3', 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', @@ -63,10 +61,9 @@ setup( 'Topic :: Software Development :: Libraries :: Application Frameworks', 'Topic :: Software Development :: Libraries :: Python Modules', ], - python_requires='>=2.7, !=3.0.*, !=3.1.*, !=3.2.*, !=3.3.*, !=3.4.*', + python_requires='>=3.5', install_requires=[ - 'Twisted>=16.0.0;python_version=="2.7"', - 'Twisted>=17.9.0;python_version>="3.5"', + 'Twisted>=17.9.0', 'cryptography>=2.0', 'cssselect>=0.9.1', 'lxml>=3.5.0', diff --git a/tests/requirements-py2.txt b/tests/requirements-py2.txt deleted file mode 100644 index f621eb4eb..000000000 --- a/tests/requirements-py2.txt +++ /dev/null @@ -1,15 +0,0 @@ -# Tests requirements -brotlipy -jmespath -mitmproxy==0.10.1 -mock -netlib==0.10.1 -pytest -pytest-cov -pytest-twisted -pytest-xdist -testfixtures - -# optional for shell wrapper tests -bpython -ipython<6.0 diff --git a/tox.ini b/tox.ini index fe925951b..821144381 100644 --- a/tox.ini +++ b/tox.ini @@ -4,18 +4,16 @@ # and then run "tox" from this directory. [tox] -envlist = py27 +envlist = py35 [testenv] deps = -ctests/constraints.txt - -rrequirements-py2.txt + -rrequirements-py3.txt + -rtests/requirements-py3.txt # Extras botocore>=1.3.23 - google-cloud-storage - leveldb Pillow>=3.4.2 - -rtests/requirements-py2.txt passenv = S3_TEST_FILE_URI AWS_ACCESS_KEY_ID @@ -25,42 +23,8 @@ passenv = commands = py.test --cov=scrapy --cov-report= {posargs:scrapy tests} -[testenv:py27-pinned] -basepython = python2.7 -deps = - -ctests/constraints.txt - cryptography==2.0 - cssselect==0.9.1 - lxml==3.5.0 - parsel==1.5.0 - Protego==0.1.15 - PyDispatcher==2.0.5 - pyOpenSSL==16.2.0 - queuelib==1.4.2 - service_identity==16.0.0 - six==1.10.0 - Twisted==16.0.0 - w3lib==1.17.0 - zope.interface==4.1.3 - -rtests/requirements-py2.txt - # Extras - botocore==1.3.23 - Pillow==3.4.2 - -[testenv:pypy] -basepython = pypy -commands = - py.test {posargs:scrapy tests} - [testenv:py35] basepython = python3.5 -deps = - -ctests/constraints.txt - -rrequirements-py3.txt - -rtests/requirements-py3.txt - # Extras - botocore>=1.3.23 - Pillow>=3.4.2 [testenv:py35-pinned] basepython = python3.5 @@ -86,19 +50,15 @@ deps = [testenv:py36] basepython = python3.6 -deps = {[testenv:py35]deps} [testenv:py37] basepython = python3.7 -deps = {[testenv:py35]deps} [testenv:py38] basepython = python3.8 -deps = {[testenv:py35]deps} [testenv:pypy3] basepython = pypy3 -deps = {[testenv:py35]deps} commands = py.test {posargs:scrapy tests} @@ -127,13 +87,6 @@ commands = [testenv:py38-extra-deps] basepython = python3.8 -deps = - {[testenv:py35]deps} - reppy - robotexclusionrulesparser - -[testenv:py27-extra-deps] -basepython = python2.7 deps = {[testenv]deps} reppy From 15c55d0c1d4a4384fd9725d47c305f52918e3831 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 31 Oct 2019 10:47:29 +0100 Subject: [PATCH 28/34] Remove LevelDB support (#4112) --- scrapy/extensions/httpcache.py | 71 -------------------- tests/requirements-py3.txt | 1 - tests/test_downloadermiddleware_httpcache.py | 9 --- 3 files changed, 81 deletions(-) diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 7c650a91e..f3fabf710 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -347,77 +347,6 @@ class FilesystemCacheStorage(object): return pickle.load(f) -class LeveldbCacheStorage(object): - - def __init__(self, settings): - warn("The LevelDB storage backend is deprecated.", - ScrapyDeprecationWarning, stacklevel=2) - import leveldb - self._leveldb = leveldb - self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True) - self.expiration_secs = settings.getint('HTTPCACHE_EXPIRATION_SECS') - self.db = None - - def open_spider(self, spider): - dbpath = os.path.join(self.cachedir, '%s.leveldb' % spider.name) - self.db = self._leveldb.LevelDB(dbpath) - - logger.debug("Using LevelDB cache storage in %(cachepath)s" % {'cachepath': dbpath}, extra={'spider': spider}) - - def close_spider(self, spider): - # Do compactation each time to save space and also recreate files to - # avoid them being removed in storages with timestamp-based autoremoval. - self.db.CompactRange() - del self.db - garbage_collect() - - def retrieve_response(self, spider, request): - data = self._read_data(spider, request) - if data is None: - return # not cached - url = data['url'] - status = data['status'] - headers = Headers(data['headers']) - body = data['body'] - respcls = responsetypes.from_args(headers=headers, url=url) - response = respcls(url=url, headers=headers, status=status, body=body) - return response - - def store_response(self, spider, request, response): - key = self._request_key(request) - data = { - 'status': response.status, - 'url': response.url, - 'headers': dict(response.headers), - 'body': response.body, - } - batch = self._leveldb.WriteBatch() - batch.Put(key + b'_data', pickle.dumps(data, protocol=2)) - batch.Put(key + b'_time', to_bytes(str(time()))) - self.db.Write(batch) - - def _read_data(self, spider, request): - key = self._request_key(request) - try: - ts = self.db.Get(key + b'_time') - except KeyError: - return # not found or invalid entry - - if 0 < self.expiration_secs < time() - float(ts): - return # expired - - try: - data = self.db.Get(key + b'_data') - except KeyError: - return # invalid entry - else: - return pickle.loads(data) - - def _request_key(self, request): - return to_bytes(request_fingerprint(request)) - - - def parse_cachecontrol(header): """Parse Cache-Control header diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index cb67bc40e..dd5b23cc3 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -1,6 +1,5 @@ # Tests requirements jmespath -leveldb; sys_platform != "win32" pytest pytest-cov pytest-twisted diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 972d400a4..950664ffe 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -156,15 +156,6 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): return super(FilesystemStorageTest, self)._get_settings(**new_settings) -class LeveldbStorageTest(DefaultStorageTest): - - try: - pytest.importorskip('leveldb') - except SystemError: - pytestmark = pytest.mark.skip("Test module skipped - 'SystemError: bad call flags' occurs when >= Python 3.8") - storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' - - class DummyPolicyTest(_BaseTest): policy_class = 'scrapy.extensions.httpcache.DummyPolicy' From f02c3d1dcf3e4880388d19e961e7911be5dc54ff Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 31 Oct 2019 13:31:33 +0100 Subject: [PATCH 29/34] Use communicate() instead of wait() after killing the mock server (#4095) --- tests/mockserver.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/mockserver.py b/tests/mockserver.py index 77908284b..b766bb653 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -206,8 +206,7 @@ class MockServer(): def __exit__(self, exc_type, exc_value, traceback): self.proc.kill() - self.proc.wait() - time.sleep(0.2) + self.proc.communicate() def url(self, path, is_secure=False): host = self.http_address.replace('0.0.0.0', '127.0.0.1') From 698aa704b98e206cb755190f74d73a6ba47c5fac Mon Sep 17 00:00:00 2001 From: seregaxvm Date: Tue, 5 Nov 2019 18:30:01 +0300 Subject: [PATCH 30/34] Fix zsh completion file extension (#4122) --- extras/scrapy_zsh_completion | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion index 86c52c36c..e995947cb 100644 --- a/extras/scrapy_zsh_completion +++ b/extras/scrapy_zsh_completion @@ -61,7 +61,7 @@ _scrapy() { '-c[evaluate the code in the shell, print the result and exit]:code:(CODE)' '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' '--spider[use this spider]:spider:_scrapy_spiders' - '::file:_files -g \*.http' + '::file:_files -g \*.html' '::URL:_httpie_urls' ) _scrapy_glb_opts $options From 98caf055b5a343ff663f2c1150b301a3f4546a16 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 6 Nov 2019 11:53:46 +0100 Subject: [PATCH 31/34] =?UTF-8?q?Fix=20a=20typo:=20specifiy=20=E2=86=92=20?= =?UTF-8?q?specify=20(#4128)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scrapy/utils/misc.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index f638adb25..b74f34451 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -136,7 +136,7 @@ def create_instance(objcls, settings, crawler, *args, **kwargs): """ if settings is None: if crawler is None: - raise ValueError("Specifiy at least one of settings and crawler.") + raise ValueError("Specify at least one of settings and crawler.") settings = crawler.settings if crawler and hasattr(objcls, 'from_crawler'): return objcls.from_crawler(crawler, *args, **kwargs) From e8b1e46e85fbcdf22408d320f3cc61b6e802c5e2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Marc=20Hern=C3=A1ndez?= Date: Thu, 7 Nov 2019 14:05:01 +0100 Subject: [PATCH 32/34] Add pytest-flake8 (#3945) --- .travis.yml | 2 + conftest.py | 10 +++ pytest.ini | 253 ++++++++++++++++++++++++++++++++++++++++++++++++++++ tox.ini | 8 ++ 4 files changed, 273 insertions(+) diff --git a/.travis.yml b/.travis.yml index 2352ef124..f5b9d3f72 100644 --- a/.travis.yml +++ b/.travis.yml @@ -7,6 +7,8 @@ branches: - /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/ matrix: include: + - env: TOXENV=flake8 + python: 3.8 - env: TOXENV=pypy3 python: 3.5 - env: TOXENV=py35 diff --git a/conftest.py b/conftest.py index 06d65ba1d..d5d61ddd3 100644 --- a/conftest.py +++ b/conftest.py @@ -19,3 +19,13 @@ if six.PY3: def chdir(tmpdir): """Change to pytest-provided temporary directory""" tmpdir.chdir() + + +def pytest_collection_modifyitems(session, config, items): + # Avoid executing tests when executing `--flake8` flag (pytest-flake8) + try: + from pytest_flake8 import Flake8Item + if config.getoption('--flake8'): + items[:] = [item for item in items if isinstance(item, Flake8Item)] + except ImportError: + pass diff --git a/pytest.ini b/pytest.ini index 73d169601..0e4ee9ed1 100644 --- a/pytest.ini +++ b/pytest.ini @@ -4,3 +4,256 @@ python_files=test_*.py __init__.py python_classes= addopts = --doctest-modules --assert=plain twisted = 1 +flake8-ignore = + # extras + extras/qps-bench-server.py E261 E501 + extras/qpsclient.py E501 E261 E501 + # scrapy/commands + scrapy/commands/__init__.py E128 E501 + scrapy/commands/check.py F401 E501 W391 + scrapy/commands/crawl.py E501 + scrapy/commands/edit.py E501 + scrapy/commands/fetch.py E401 E302 E501 E128 E502 E731 + scrapy/commands/genspider.py E128 E501 E502 + scrapy/commands/list.py E302 + scrapy/commands/parse.py E128 E501 E731 E226 + scrapy/commands/runspider.py E501 + scrapy/commands/settings.py E302 E128 + scrapy/commands/shell.py E128 E501 E502 + scrapy/commands/startproject.py E502 E127 E501 E128 W391 + scrapy/commands/version.py E501 E128 W391 + scrapy/commands/view.py F401 E302 + # scrapy/contracts + scrapy/contracts/__init__.py E501 W504 + scrapy/contracts/default.py E502 E128 + # scrapy/core + scrapy/core/engine.py E261 E501 E128 E127 E306 E502 + scrapy/core/scheduler.py E501 + scrapy/core/scraper.py E501 E306 E261 E128 W391 W504 + scrapy/core/spidermw.py E501 E731 E502 E231 E126 E226 + scrapy/core/downloader/__init__.py F401 E501 + scrapy/core/downloader/contextfactory.py E501 E128 E126 + scrapy/core/downloader/middleware.py E501 E502 + scrapy/core/downloader/tls.py E501 E305 E241 + scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 W391 E126 E226 + scrapy/core/downloader/handlers/__init__.py E501 + scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 W391 + scrapy/core/downloader/handlers/http.py F401 + scrapy/core/downloader/handlers/http10.py E501 + scrapy/core/downloader/handlers/http11.py E501 + scrapy/core/downloader/handlers/s3.py E501 F401 E502 E128 E126 + # scrapy/downloadermiddlewares + scrapy/downloadermiddlewares/ajaxcrawl.py E302 E501 E226 + scrapy/downloadermiddlewares/decompression.py E501 + scrapy/downloadermiddlewares/defaultheaders.py E501 + scrapy/downloadermiddlewares/httpcache.py E501 E126 + scrapy/downloadermiddlewares/httpcompression.py E502 E128 + scrapy/downloadermiddlewares/httpproxy.py E501 + scrapy/downloadermiddlewares/redirect.py E501 W504 + scrapy/downloadermiddlewares/retry.py E501 E126 + scrapy/downloadermiddlewares/robotstxt.py F401 E501 + scrapy/downloadermiddlewares/stats.py E501 + # scrapy/extensions + scrapy/extensions/closespider.py E501 E502 E128 E123 + scrapy/extensions/corestats.py E302 E501 + scrapy/extensions/feedexport.py E128 E501 + scrapy/extensions/httpcache.py E128 E501 E303 F401 + scrapy/extensions/memdebug.py E501 + scrapy/extensions/spiderstate.py E302 E501 + scrapy/extensions/telnet.py E501 W504 + scrapy/extensions/throttle.py E501 + # scrapy/http + scrapy/http/__init__.py F401 + scrapy/http/common.py E501 + scrapy/http/cookies.py E501 + scrapy/http/headers.py W391 + scrapy/http/request/__init__.py E501 + scrapy/http/request/form.py E501 E123 + scrapy/http/request/json_request.py E501 + scrapy/http/response/__init__.py E501 E128 W293 W291 + scrapy/http/response/html.py E302 + scrapy/http/response/text.py E501 W293 E128 E124 + scrapy/http/response/xml.py E302 + # scrapy/linkextractors + scrapy/linkextractors/__init__.py E731 E502 E501 E402 F401 + scrapy/linkextractors/lxmlhtml.py E501 E731 E226 + # scrapy/loader + scrapy/loader/__init__.py E501 E502 E128 + scrapy/loader/common.py E302 + scrapy/loader/processors.py E501 + # scrapy/pipelines + scrapy/pipelines/__init__.py E302 + scrapy/pipelines/files.py E116 E501 E266 + scrapy/pipelines/images.py E265 E501 + scrapy/pipelines/media.py E125 E501 E266 + # scrapy/selector + scrapy/selector/__init__.py F403 F401 + scrapy/selector/unified.py F401 E501 E111 + # scrapy/settings + scrapy/settings/__init__.py E501 + scrapy/settings/default_settings.py E501 E261 E114 E116 E226 + scrapy/settings/deprecated.py E501 + # scrapy/spidermiddlewares + scrapy/spidermiddlewares/httperror.py E501 + scrapy/spidermiddlewares/offsite.py E501 + scrapy/spidermiddlewares/referer.py F401 E501 E129 W503 W504 + scrapy/spidermiddlewares/urllength.py E501 + # scrapy/spiders + scrapy/spiders/__init__.py F401 E501 E402 + scrapy/spiders/crawl.py E501 + scrapy/spiders/feed.py E501 E261 W391 + scrapy/spiders/init.py W391 + scrapy/spiders/sitemap.py E501 + # scrapy/utils + scrapy/utils/benchserver.py E501 + scrapy/utils/boto.py F401 + scrapy/utils/conf.py E402 E502 E501 + scrapy/utils/console.py E302 E261 F401 E306 E305 + scrapy/utils/curl.py F401 + scrapy/utils/datatypes.py E501 E226 + scrapy/utils/decorators.py E501 E302 + scrapy/utils/defer.py E501 E302 E128 + scrapy/utils/deprecate.py E128 E501 E127 E502 + scrapy/utils/display.py E302 + scrapy/utils/engine.py F401 E261 E302 + scrapy/utils/ftp.py E302 + scrapy/utils/gz.py E305 E501 E302 W504 + scrapy/utils/http.py F403 F401 W391 E226 + scrapy/utils/httpobj.py E302 E501 + scrapy/utils/iterators.py E501 E701 + scrapy/utils/job.py E302 + scrapy/utils/log.py E128 W503 + scrapy/utils/markup.py F403 F401 W292 + scrapy/utils/misc.py E501 E226 + scrapy/utils/multipart.py F403 F401 W292 + scrapy/utils/project.py E501 + scrapy/utils/python.py E501 E302 + scrapy/utils/reactor.py E302 E226 + scrapy/utils/reqser.py E501 + scrapy/utils/request.py E302 E127 E501 + scrapy/utils/response.py E501 E302 E128 + scrapy/utils/signal.py E501 E128 + scrapy/utils/sitemap.py E501 + scrapy/utils/spider.py E271 E302 E501 + scrapy/utils/ssl.py E501 + scrapy/utils/template.py E302 + scrapy/utils/test.py E302 E501 + scrapy/utils/url.py E501 F403 F401 E128 F405 + # scrapy + scrapy/__init__.py E402 E501 + scrapy/_monkeypatches.py W293 + scrapy/cmdline.py E502 E501 + scrapy/crawler.py E501 + scrapy/dupefilters.py E302 E501 E202 + scrapy/exceptions.py E302 E501 + scrapy/exporters.py E501 E261 E226 + scrapy/extension.py E302 + scrapy/interfaces.py E302 E501 W391 + scrapy/item.py E501 E128 + scrapy/link.py E501 W391 + scrapy/logformatter.py E501 W293 + scrapy/mail.py E402 E128 E501 E502 + scrapy/middleware.py E502 E128 E501 + scrapy/pqueues.py E501 + scrapy/resolver.py E302 + scrapy/responsetypes.py E128 E501 E305 + scrapy/robotstxt.py E302 E501 + scrapy/shell.py E501 + scrapy/signalmanager.py E501 + scrapy/spiderloader.py E225 F841 E501 E126 + scrapy/squeues.py E128 + scrapy/statscollectors.py E501 W391 + # tests + tests/__init__.py F401 E402 E501 + tests/mockserver.py E401 E501 E126 E123 F401 + tests/pipelines.py E302 F841 E226 + tests/spiders.py E302 E501 E127 + tests/test_closespider.py E501 E127 + tests/test_command_fetch.py E501 E261 + tests/test_command_parse.py F401 E302 E501 E128 E303 E226 + tests/test_command_shell.py E501 E128 + tests/test_commands.py F401 E128 E501 + tests/test_contracts.py E501 E128 W293 + tests/test_crawl.py E501 E741 E265 + tests/test_crawler.py F841 E306 E501 + tests/test_dependencies.py E302 F841 E501 E305 + tests/test_downloader_handlers.py E124 E127 E128 E225 E261 E265 F401 E501 E502 E701 E711 E126 E226 E123 + tests/test_downloadermiddleware.py E501 + tests/test_downloadermiddleware_ajaxcrawlable.py E302 E501 + tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126 + tests/test_downloadermiddleware_decompression.py E127 + tests/test_downloadermiddleware_defaultheaders.py E501 + tests/test_downloadermiddleware_downloadtimeout.py E501 + tests/test_downloadermiddleware_httpcache.py E713 E501 E302 E305 F401 + tests/test_downloadermiddleware_httpcompression.py E501 F401 E251 E126 E123 + tests/test_downloadermiddleware_httpproxy.py F401 E501 E128 + tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 + tests/test_downloadermiddleware_retry.py E501 E128 W293 E251 E502 E303 E126 + tests/test_downloadermiddleware_robotstxt.py E501 + tests/test_downloadermiddleware_stats.py E501 + tests/test_dupefilters.py E302 E221 E501 E741 W293 W291 E128 E124 + tests/test_engine.py E401 E501 E502 E128 E261 + tests/test_exporters.py E501 E731 E306 E128 E124 + tests/test_extension_telnet.py F401 F841 + tests/test_feedexport.py E501 F401 F841 E241 + tests/test_http_cookies.py E501 + tests/test_http_headers.py E302 E501 + tests/test_http_request.py F401 E402 E501 E231 E261 E127 E128 W293 E502 E128 E502 E126 E123 + tests/test_http_response.py E501 E301 E502 E128 E265 + tests/test_item.py E701 E128 E231 F841 E306 + tests/test_link.py E501 + tests/test_linkextractors.py E501 E128 E231 E124 + tests/test_loader.py E302 E501 E731 E303 E741 E128 E117 E241 + tests/test_logformatter.py E128 E501 E231 E122 E302 + tests/test_mail.py E302 E128 E501 E305 + tests/test_middleware.py E302 E501 E128 + tests/test_pipeline_crawl.py E131 E501 E128 E126 + tests/test_pipeline_files.py F401 E501 W293 E303 E272 E226 + tests/test_pipeline_images.py F401 F841 E501 E303 + tests/test_pipeline_media.py E501 E741 E731 E128 E261 E306 E502 + tests/test_request_cb_kwargs.py E501 + tests/test_responsetypes.py E501 E302 E305 + tests/test_robotstxt_interface.py F401 E302 E501 W291 E501 + tests/test_scheduler.py E501 E126 E123 + tests/test_selector.py F401 E501 E127 + tests/test_spider.py E501 F401 + tests/test_spidermiddleware.py E501 E226 + tests/test_spidermiddleware_depth.py W391 + tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 + tests/test_spidermiddleware_offsite.py E302 E501 E128 E111 W293 + tests/test_spidermiddleware_output_chain.py F401 E501 E302 W293 E226 + tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 W391 E241 E121 + tests/test_spidermiddleware_urllength.py W391 + tests/test_squeues.py E501 E302 E701 E741 + tests/test_utils_conf.py E501 E231 E303 E128 + tests/test_utils_console.py E302 E231 + tests/test_utils_curl.py E501 + tests/test_utils_datatypes.py E402 E501 E305 W391 + tests/test_utils_defer.py E306 E261 E501 E302 F841 E226 + tests/test_utils_deprecate.py F841 E306 E501 + tests/test_utils_http.py E302 E501 E502 E128 W391 W504 + tests/test_utils_httpobj.py E302 + tests/test_utils_iterators.py E501 E128 E129 E302 E303 E241 + tests/test_utils_log.py E741 E226 + tests/test_utils_python.py E501 E303 E731 E701 E305 + tests/test_utils_reqser.py F401 E501 E128 + tests/test_utils_request.py E302 E501 E128 E305 + tests/test_utils_response.py E501 + tests/test_utils_signal.py E741 F841 E302 E731 E226 + tests/test_utils_sitemap.py E302 E128 E501 E124 + tests/test_utils_spider.py E261 E302 E305 W391 + tests/test_utils_template.py E305 + tests/test_utils_url.py F401 E501 E127 E302 E305 E211 E125 E501 E226 E241 E126 E123 + tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 + tests/mocks/dummydbm.py E302 + tests/test_cmdline/__init__.py E502 E501 + tests/test_cmdline/extensions.py E302 W391 + tests/test_settings/__init__.py F401 E501 E128 + tests/test_settings/default_settings.py W391 + tests/test_spiderloader/__init__.py E128 E501 E302 + tests/test_spiderloader/test_spiders/spider0.py E302 + tests/test_spiderloader/test_spiders/spider1.py E302 + tests/test_spiderloader/test_spiders/spider2.py E302 + tests/test_spiderloader/test_spiders/spider3.py E302 + tests/test_spiderloader/test_spiders/nested/spider4.py E302 + tests/test_utils_misc/__init__.py E501 E231 diff --git a/tox.ini b/tox.ini index 821144381..cc3463a4d 100644 --- a/tox.ini +++ b/tox.ini @@ -62,6 +62,14 @@ basepython = pypy3 commands = py.test {posargs:scrapy tests} +[testenv:flake8] +basepython = python3.8 +deps = + {[testenv]deps} + pytest-flake8 +commands = + py.test --flake8 {posargs:scrapy tests} + [docs] changedir = docs deps = From c377c14e3263ce3c2bffa446cd0965006e845664 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Marc=20Hern=C3=A1ndez?= Date: Thu, 7 Nov 2019 17:47:35 +0100 Subject: [PATCH 33/34] Fix W391 Blank line at end of file (#4137) --- pytest.ini | 37 ++++++++++-------------- scrapy/commands/check.py | 1 - scrapy/commands/startproject.py | 1 - scrapy/commands/version.py | 1 - scrapy/core/downloader/handlers/ftp.py | 1 - scrapy/core/downloader/webclient.py | 1 - scrapy/core/scraper.py | 1 - scrapy/http/headers.py | 2 -- scrapy/interfaces.py | 1 - scrapy/link.py | 1 - scrapy/spiders/feed.py | 1 - scrapy/spiders/init.py | 1 - scrapy/statscollectors.py | 2 -- scrapy/utils/http.py | 1 - tests/test_cmdline/extensions.py | 1 - tests/test_settings/default_settings.py | 1 - tests/test_spidermiddleware_depth.py | 1 - tests/test_spidermiddleware_urllength.py | 1 - tests/test_utils_datatypes.py | 1 - tests/test_utils_http.py | 2 -- tests/test_utils_spider.py | 1 - 21 files changed, 16 insertions(+), 44 deletions(-) diff --git a/pytest.ini b/pytest.ini index 0e4ee9ed1..db5bee228 100644 --- a/pytest.ini +++ b/pytest.ini @@ -10,7 +10,7 @@ flake8-ignore = extras/qpsclient.py E501 E261 E501 # scrapy/commands scrapy/commands/__init__.py E128 E501 - scrapy/commands/check.py F401 E501 W391 + scrapy/commands/check.py F401 E501 scrapy/commands/crawl.py E501 scrapy/commands/edit.py E501 scrapy/commands/fetch.py E401 E302 E501 E128 E502 E731 @@ -20,8 +20,8 @@ flake8-ignore = scrapy/commands/runspider.py E501 scrapy/commands/settings.py E302 E128 scrapy/commands/shell.py E128 E501 E502 - scrapy/commands/startproject.py E502 E127 E501 E128 W391 - scrapy/commands/version.py E501 E128 W391 + scrapy/commands/startproject.py E502 E127 E501 E128 + scrapy/commands/version.py E501 E128 scrapy/commands/view.py F401 E302 # scrapy/contracts scrapy/contracts/__init__.py E501 W504 @@ -29,15 +29,15 @@ flake8-ignore = # scrapy/core scrapy/core/engine.py E261 E501 E128 E127 E306 E502 scrapy/core/scheduler.py E501 - scrapy/core/scraper.py E501 E306 E261 E128 W391 W504 + scrapy/core/scraper.py E501 E306 E261 E128 W504 scrapy/core/spidermw.py E501 E731 E502 E231 E126 E226 scrapy/core/downloader/__init__.py F401 E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 scrapy/core/downloader/tls.py E501 E305 E241 - scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 W391 E126 E226 + scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 E126 E226 scrapy/core/downloader/handlers/__init__.py E501 - scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 W391 + scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 scrapy/core/downloader/handlers/http.py F401 scrapy/core/downloader/handlers/http10.py E501 scrapy/core/downloader/handlers/http11.py E501 @@ -66,7 +66,6 @@ flake8-ignore = scrapy/http/__init__.py F401 scrapy/http/common.py E501 scrapy/http/cookies.py E501 - scrapy/http/headers.py W391 scrapy/http/request/__init__.py E501 scrapy/http/request/form.py E501 E123 scrapy/http/request/json_request.py E501 @@ -101,8 +100,7 @@ flake8-ignore = # scrapy/spiders scrapy/spiders/__init__.py F401 E501 E402 scrapy/spiders/crawl.py E501 - scrapy/spiders/feed.py E501 E261 W391 - scrapy/spiders/init.py W391 + scrapy/spiders/feed.py E501 E261 scrapy/spiders/sitemap.py E501 # scrapy/utils scrapy/utils/benchserver.py E501 @@ -118,7 +116,7 @@ flake8-ignore = scrapy/utils/engine.py F401 E261 E302 scrapy/utils/ftp.py E302 scrapy/utils/gz.py E305 E501 E302 W504 - scrapy/utils/http.py F403 F401 W391 E226 + scrapy/utils/http.py F403 F401 E226 scrapy/utils/httpobj.py E302 E501 scrapy/utils/iterators.py E501 E701 scrapy/utils/job.py E302 @@ -148,9 +146,9 @@ flake8-ignore = scrapy/exceptions.py E302 E501 scrapy/exporters.py E501 E261 E226 scrapy/extension.py E302 - scrapy/interfaces.py E302 E501 W391 + scrapy/interfaces.py E302 E501 scrapy/item.py E501 E128 - scrapy/link.py E501 W391 + scrapy/link.py E501 scrapy/logformatter.py E501 W293 scrapy/mail.py E402 E128 E501 E502 scrapy/middleware.py E502 E128 E501 @@ -162,7 +160,7 @@ flake8-ignore = scrapy/signalmanager.py E501 scrapy/spiderloader.py E225 F841 E501 E126 scrapy/squeues.py E128 - scrapy/statscollectors.py E501 W391 + scrapy/statscollectors.py E501 # tests tests/__init__.py F401 E402 E501 tests/mockserver.py E401 E501 E126 E123 F401 @@ -218,20 +216,18 @@ flake8-ignore = tests/test_selector.py F401 E501 E127 tests/test_spider.py E501 F401 tests/test_spidermiddleware.py E501 E226 - tests/test_spidermiddleware_depth.py W391 tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 tests/test_spidermiddleware_offsite.py E302 E501 E128 E111 W293 tests/test_spidermiddleware_output_chain.py F401 E501 E302 W293 E226 - tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 W391 E241 E121 - tests/test_spidermiddleware_urllength.py W391 + tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 E241 E121 tests/test_squeues.py E501 E302 E701 E741 tests/test_utils_conf.py E501 E231 E303 E128 tests/test_utils_console.py E302 E231 tests/test_utils_curl.py E501 - tests/test_utils_datatypes.py E402 E501 E305 W391 + tests/test_utils_datatypes.py E402 E501 E305 tests/test_utils_defer.py E306 E261 E501 E302 F841 E226 tests/test_utils_deprecate.py F841 E306 E501 - tests/test_utils_http.py E302 E501 E502 E128 W391 W504 + tests/test_utils_http.py E302 E501 E502 E128 W504 tests/test_utils_httpobj.py E302 tests/test_utils_iterators.py E501 E128 E129 E302 E303 E241 tests/test_utils_log.py E741 E226 @@ -241,15 +237,14 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 E302 E731 E226 tests/test_utils_sitemap.py E302 E128 E501 E124 - tests/test_utils_spider.py E261 E302 E305 W391 + tests/test_utils_spider.py E261 E302 E305 tests/test_utils_template.py E305 tests/test_utils_url.py F401 E501 E127 E302 E305 E211 E125 E501 E226 E241 E126 E123 tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 tests/mocks/dummydbm.py E302 tests/test_cmdline/__init__.py E502 E501 - tests/test_cmdline/extensions.py E302 W391 + tests/test_cmdline/extensions.py E302 tests/test_settings/__init__.py F401 E501 E128 - tests/test_settings/default_settings.py W391 tests/test_spiderloader/__init__.py E128 E501 E302 tests/test_spiderloader/test_spiders/spider0.py E302 tests/test_spiderloader/test_spiders/spider1.py E302 diff --git a/scrapy/commands/check.py b/scrapy/commands/check.py index ab73e85e7..3e6c11b7d 100644 --- a/scrapy/commands/check.py +++ b/scrapy/commands/check.py @@ -96,4 +96,3 @@ class Command(ScrapyCommand): result.printErrors() result.printSummary(start, stop) self.exitcode = int(not result.wasSuccessful()) - diff --git a/scrapy/commands/startproject.py b/scrapy/commands/startproject.py index 67337c26e..3b9f6eabb 100644 --- a/scrapy/commands/startproject.py +++ b/scrapy/commands/startproject.py @@ -119,4 +119,3 @@ class Command(ScrapyCommand): _templates_base_dir = self.settings['TEMPLATES_DIR'] or \ join(scrapy.__path__[0], 'templates') return join(_templates_base_dir, 'project') - diff --git a/scrapy/commands/version.py b/scrapy/commands/version.py index 577365c3b..8651948f7 100644 --- a/scrapy/commands/version.py +++ b/scrapy/commands/version.py @@ -30,4 +30,3 @@ class Command(ScrapyCommand): print(patt % (name, version)) else: print("Scrapy %s" % scrapy.__version__) - diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 806a537d4..39ed67a1a 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -112,4 +112,3 @@ class FTPDownloadHandler(object): httpcode = self.CODE_MAPPING.get(ftpcode, self.CODE_MAPPING["default"]) return Response(url=request.url, status=httpcode, body=to_bytes(message)) raise result.type(result.value) - diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index 3a5890ed0..3fe13414a 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -157,4 +157,3 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): def gotHeaders(self, headers): self.headers_time = time() self.response_headers = headers - diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 1f389cf2e..40de6b87a 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -244,4 +244,3 @@ class Scraper(object): return self.signals.send_catch_log_deferred( signal=signals.item_scraped, item=output, response=response, spider=spider) - diff --git a/scrapy/http/headers.py b/scrapy/http/headers.py index 62507eb19..f3b46b994 100644 --- a/scrapy/http/headers.py +++ b/scrapy/http/headers.py @@ -91,5 +91,3 @@ class Headers(CaselessDict): def __copy__(self): return self.__class__(self) copy = __copy__ - - diff --git a/scrapy/interfaces.py b/scrapy/interfaces.py index 89ad2b14f..d48babc3c 100644 --- a/scrapy/interfaces.py +++ b/scrapy/interfaces.py @@ -15,4 +15,3 @@ class ISpiderLoader(Interface): def find_by_request(request): """Return the list of spiders names that can handle the given request""" - diff --git a/scrapy/link.py b/scrapy/link.py index 2c8301680..f0638ced2 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -39,4 +39,3 @@ class Link(object): def __repr__(self): return 'Link(url=%r, text=%r, fragment=%r, nofollow=%r)' % \ (self.url, self.text, self.fragment, self.nofollow) - diff --git a/scrapy/spiders/feed.py b/scrapy/spiders/feed.py index 06e212e1c..197812a26 100644 --- a/scrapy/spiders/feed.py +++ b/scrapy/spiders/feed.py @@ -133,4 +133,3 @@ class CSVFeedSpider(Spider): raise NotConfigured('You must define parse_row method in order to scrape this CSV feed') response = self.adapt_response(response) return self.parse_rows(response) - diff --git a/scrapy/spiders/init.py b/scrapy/spiders/init.py index 2efb1a869..fd41133ea 100644 --- a/scrapy/spiders/init.py +++ b/scrapy/spiders/init.py @@ -29,4 +29,3 @@ class InitSpider(Spider): spider """ return self.initialized() - diff --git a/scrapy/statscollectors.py b/scrapy/statscollectors.py index 6da9ddcd2..f0bfaed34 100644 --- a/scrapy/statscollectors.py +++ b/scrapy/statscollectors.py @@ -80,5 +80,3 @@ class DummyStatsCollector(StatsCollector): def min_value(self, key, value, spider=None): pass - - diff --git a/scrapy/utils/http.py b/scrapy/utils/http.py index b6e05c862..ad49ef3e9 100644 --- a/scrapy/utils/http.py +++ b/scrapy/utils/http.py @@ -34,4 +34,3 @@ def decode_chunked_transfer(chunked_body): body += t[:size] t = t[size+2:] return body - diff --git a/tests/test_cmdline/extensions.py b/tests/test_cmdline/extensions.py index 72867eb56..28456b55d 100644 --- a/tests/test_cmdline/extensions.py +++ b/tests/test_cmdline/extensions.py @@ -12,4 +12,3 @@ class TestExtension(object): class DummyExtension(object): pass - diff --git a/tests/test_settings/default_settings.py b/tests/test_settings/default_settings.py index c24b5a9b9..26a555275 100644 --- a/tests/test_settings/default_settings.py +++ b/tests/test_settings/default_settings.py @@ -2,4 +2,3 @@ TEST_DEFAULT = 'defvalue' TEST_DICT = {'key': 'val'} - diff --git a/tests/test_spidermiddleware_depth.py b/tests/test_spidermiddleware_depth.py index 3685d5a6f..71cca2472 100644 --- a/tests/test_spidermiddleware_depth.py +++ b/tests/test_spidermiddleware_depth.py @@ -40,4 +40,3 @@ class TestDepthMiddleware(TestCase): def tearDown(self): self.stats.close_spider(self.spider, '') - diff --git a/tests/test_spidermiddleware_urllength.py b/tests/test_spidermiddleware_urllength.py index a0aae0fdd..5ef2b23fd 100644 --- a/tests/test_spidermiddleware_urllength.py +++ b/tests/test_spidermiddleware_urllength.py @@ -18,4 +18,3 @@ class TestUrlLengthMiddleware(TestCase): spider = Spider('foo') out = list(mw.process_spider_output(res, reqs, spider)) self.assertEqual(out, [short_url_req]) - diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 535095b8d..9455172fa 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -244,4 +244,3 @@ class SequenceExcludeTest(unittest.TestCase): if __name__ == "__main__": unittest.main() - diff --git a/tests/test_utils_http.py b/tests/test_utils_http.py index 583105673..2524153ea 100644 --- a/tests/test_utils_http.py +++ b/tests/test_utils_http.py @@ -16,5 +16,3 @@ class ChunkedTest(unittest.TestCase): "This is the data in the first chunk\r\n" + "and this is the second one\r\n" + "consequence") - - diff --git a/tests/test_utils_spider.py b/tests/test_utils_spider.py index 045e72117..d9de1ce77 100644 --- a/tests/test_utils_spider.py +++ b/tests/test_utils_spider.py @@ -34,4 +34,3 @@ class UtilsSpidersTestCase(unittest.TestCase): if __name__ == "__main__": unittest.main() - From d874c4d90bcf96c7e5b507babaa2a45a233da506 Mon Sep 17 00:00:00 2001 From: Andrey Rahmatullin Date: Thu, 7 Nov 2019 22:02:17 +0500 Subject: [PATCH 34/34] Remove the old Python 2 PyPy installation code from .travis.yml (#4138) --- .travis.yml | 7 ------- 1 file changed, 7 deletions(-) diff --git a/.travis.yml b/.travis.yml index f5b9d3f72..0e77af9fd 100644 --- a/.travis.yml +++ b/.travis.yml @@ -27,13 +27,6 @@ matrix: python: 3.6 install: - | - if [ "$TOXENV" = "pypy" ]; then - export PYPY_VERSION="pypy-6.0.0-linux_x86_64-portable" - wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2" - tar -jxf ${PYPY_VERSION}.tar.bz2 - virtualenv --python="$PYPY_VERSION/bin/pypy" "$HOME/virtualenvs/$PYPY_VERSION" - source "$HOME/virtualenvs/$PYPY_VERSION/bin/activate" - fi if [ "$TOXENV" = "pypy3" ]; then export PYPY_VERSION="pypy3.5-5.9-beta-linux_x86_64-portable" wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2"