From 18d0affc015cd1be02cc9b227c4d89007dd8c816 Mon Sep 17 00:00:00 2001 From: Shivam Sandbhor Date: Mon, 5 Aug 2019 16:53:35 +0530 Subject: [PATCH 001/113] Update reactor.py, updated 'if' sequencing , possibly eliminating a bug if portrange=None This should be the proper ordering.This is the explanation. If 'not portrange' is True ,it is guaranteed that `not hasattr(portrange, '__iter__')` is also True the converse of this is not always true.(for example, consider portrange=None, for such case we were executing the logic for `not hasattr(portrange, '__iter__')` . ).Such case is eliminated by this PR. --- scrapy/utils/reactor.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 83186a372..eda7867e3 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -3,10 +3,10 @@ from twisted.internet import reactor, error def listen_tcp(portrange, host, factory): """Like reactor.listenTCP but tries different ports in a range.""" assert len(portrange) <= 2, "invalid portrange: %s" % portrange - if not hasattr(portrange, '__iter__'): - return reactor.listenTCP(portrange, factory, interface=host) if not portrange: return reactor.listenTCP(0, factory, interface=host) + if not hasattr(portrange, '__iter__'): + return reactor.listenTCP(portrange, factory, interface=host) if len(portrange) == 1: return reactor.listenTCP(portrange[0], factory, interface=host) for x in range(portrange[0], portrange[1]+1): From 50c4cafe0ccc53064f38654fd7426ded876469f7 Mon Sep 17 00:00:00 2001 From: Tobias Hernstig Date: Sun, 11 Mar 2018 11:46:07 +0100 Subject: [PATCH 002/113] Update documentation for logging manually Usage of basicConfig() together with crawlerRunner is not recommended. Update documentation to highlight this fact. Closes #2149, #2352, #3146 --- docs/topics/logging.rst | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index 87ea43c7d..2fd85196d 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -254,18 +254,18 @@ scrapy.utils.log module when running custom scripts using :class:`~scrapy.crawler.CrawlerRunner`. In that case, its usage is not required but it's recommended. - If you plan on configuring the handlers yourself is still recommended you - call this function, passing ``install_root_handler=False``. Bear in mind - there won't be any log output set by default in that case. + Another option when running custom scripts is to manually configure the logging. + To do this you can use `logging.basicConfig()`_ to set a basic root handler. - To get you started on manually configuring logging's output, you can use - `logging.basicConfig()`_ to set a basic root handler. This is an example - on how to redirect ``INFO`` or higher messages to a file:: + Note that ``scrapy.crawler.CrawlerProcess`` automatically calls ``configure_logging``, + so it is recommended to only use `logging.basicConfig()`_ together with + ``scrapy.crawler.CrawlerRunner`` + + This is an example on how to redirect ``INFO`` or higher messages to a file:: import logging from scrapy.utils.log import configure_logging - configure_logging(install_root_handler=False) logging.basicConfig( filename='log.txt', format='%(levelname)s: %(message)s', From 2b0de0606c3b1a379722e916e9ce350e8de08a20 Mon Sep 17 00:00:00 2001 From: Tobias Hernstig Date: Thu, 15 Aug 2019 18:54:28 +0200 Subject: [PATCH 003/113] Fix merge conflicts --- docs/topics/logging.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index 2fd85196d..87ab6e19a 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -257,9 +257,9 @@ scrapy.utils.log module Another option when running custom scripts is to manually configure the logging. To do this you can use `logging.basicConfig()`_ to set a basic root handler. - Note that ``scrapy.crawler.CrawlerProcess`` automatically calls ``configure_logging``, + Note that :class:`~scrapy.crawler.CrawlerProcess` automatically calls ``configure_logging``, so it is recommended to only use `logging.basicConfig()`_ together with - ``scrapy.crawler.CrawlerRunner`` + :class:`~scrapy.crawler.CrawlerRunner`. This is an example on how to redirect ``INFO`` or higher messages to a file:: From f6872189b96595625e428331ddd4d1c620047642 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Thu, 29 Aug 2019 10:38:49 -0300 Subject: [PATCH 004/113] Add LogFormatter.error method --- scrapy/core/scraper.py | 6 +++--- scrapy/logformatter.py | 11 +++++++++++ 2 files changed, 14 insertions(+), 3 deletions(-) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 1f389cf2e..3273a1506 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -231,9 +231,9 @@ class Scraper(object): signal=signals.item_dropped, item=item, response=response, spider=spider, exception=output.value) else: - logger.error('Error processing %(item)s', {'item': item}, - exc_info=failure_to_exc_info(output), - extra={'spider': spider}) + logkws = self.logformatter.error(item, ex, response, spider) + logger.log(*logformatter_adapter(logkws), extra={'spider': spider}, + exc_info=failure_to_exc_info(output)) return self.signals.send_catch_log_deferred( signal=signals.item_error, item=item, response=response, spider=spider, failure=output) diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index f15940ed1..4437d1106 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -8,6 +8,7 @@ from scrapy.utils.request import referer_str SCRAPEDMSG = u"Scraped from %(src)s" + os.linesep + "%(item)s" DROPPEDMSG = u"Dropped: %(exception)s" + os.linesep + "%(item)s" CRAWLEDMSG = u"Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" +ERRORMSG = u"'Error processing %(item)s'" class LogFormatter(object): @@ -92,6 +93,16 @@ class LogFormatter(object): } } + def error(self, item, exception, response, spider): + """Logs a message when an item causes an error while it is passing through the item pipeline.""" + return { + 'level': logging.ERROR, + 'msg': ERRORMSG, + 'args': { + 'item': item, + } + } + @classmethod def from_crawler(cls, crawler): return cls() From 27436cbbc9e7331d18d4b63f3c894e6621226efb Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Thu, 29 Aug 2019 13:51:42 -0300 Subject: [PATCH 005/113] [test] LogFormatter.error --- tests/test_logformatter.py | 30 ++++++++++++++++++++++-------- 1 file changed, 22 insertions(+), 8 deletions(-) diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index eb9c4a561..502bc4ccc 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -23,7 +23,7 @@ class CustomItem(Item): return "name: %s" % self['name'] -class LoggingContribTest(unittest.TestCase): +class LoggingFormatterTest(unittest.TestCase): def setUp(self): self.formatter = LogFormatter() @@ -61,6 +61,16 @@ class LoggingContribTest(unittest.TestCase): lines = logline.splitlines() assert all(isinstance(x, six.text_type) for x in lines) self.assertEqual(lines, [u"Dropped: \u2018", '{}']) + + def test_error(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + item = {'key': 'value'} + exception = Exception() + response = Response("http://www.example.com") + logkws = self.formatter.error(item, exception, response, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, u"'Error processing {'key': 'value'}'") def test_scraped(self): item = CustomItem() @@ -75,26 +85,30 @@ class LoggingContribTest(unittest.TestCase): class LogFormatterSubclass(LogFormatter): def crawled(self, request, response, spider): - kwargs = super(LogFormatterSubclass, self).crawled( - request, response, spider) + kwargs = super(LogFormatterSubclass, self).crawled(request, response, spider) CRAWLEDMSG = ( - u"Crawled (%(status)s) %(request)s (referer: " - u"%(referer)s)%(flags)s" + u"Crawled (%(status)s) %(request)s (referer: %(referer)s) %(flags)s" ) + log_args = kwargs['args'] + log_args['flags'] = str(request.flags) return { 'level': kwargs['level'], 'msg': CRAWLEDMSG, - 'args': kwargs['args'] + 'args': log_args, } -class LogformatterSubclassTest(LoggingContribTest): +class LogformatterSubclassTest(unittest.TestCase): def setUp(self): self.formatter = LogFormatterSubclass() self.spider = Spider('default') def test_flags_in_request(self): - pass + req = Request("http://www.example.com", flags=['test','flag']) + res = Response("http://www.example.com") + logkws = self.formatter.crawled(req, res, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, "Crawled (200) (referer: None) ['test', 'flag']") class SkipMessagesLogFormatter(LogFormatter): From 07a31b13db96b2aef3074157808f7ad300780c3d Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 1 Oct 2019 17:55:57 -0300 Subject: [PATCH 006/113] Update LogFormatter tests --- tests/test_logformatter.py | 23 ++++++++++++++++++++--- 1 file changed, 20 insertions(+), 3 deletions(-) diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index 502bc4ccc..f12ffc11b 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -23,13 +23,13 @@ class CustomItem(Item): return "name: %s" % self['name'] -class LoggingFormatterTest(unittest.TestCase): +class LogFormatterTestCase(unittest.TestCase): def setUp(self): self.formatter = LogFormatter() self.spider = Spider('default') - def test_crawled(self): + def test_crawled_with_referer(self): req = Request("http://www.example.com") res = Response("http://www.example.com") logkws = self.formatter.crawled(req, res, self.spider) @@ -37,6 +37,7 @@ class LoggingFormatterTest(unittest.TestCase): self.assertEqual(logline, "Crawled (200) (referer: None)") + def test_crawled_without_referer(self): req = Request("http://www.example.com", headers={'referer': 'http://example.com'}) res = Response("http://www.example.com", flags=['cached']) logkws = self.formatter.crawled(req, res, self.spider) @@ -98,11 +99,27 @@ class LogFormatterSubclass(LogFormatter): } -class LogformatterSubclassTest(unittest.TestCase): +class LogformatterSubclassTest(LogFormatterTestCase): def setUp(self): self.formatter = LogFormatterSubclass() self.spider = Spider('default') + def test_crawled_with_referer(self): + req = Request("http://www.example.com") + res = Response("http://www.example.com") + logkws = self.formatter.crawled(req, res, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, + "Crawled (200) (referer: None) []") + + def test_crawled_without_referer(self): + req = Request("http://www.example.com", headers={'referer': 'http://example.com'}, flags=['cached']) + res = Response("http://www.example.com") + logkws = self.formatter.crawled(req, res, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, + "Crawled (200) (referer: http://example.com) ['cached']") + def test_flags_in_request(self): req = Request("http://www.example.com", flags=['test','flag']) res = Response("http://www.example.com") From a4aa5b8926153560c493fa84ed1d1d6c730d6f5b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 2 Oct 2019 14:51:06 +0200 Subject: [PATCH 007/113] Fix internal links in the tutorial and release notes --- docs/intro/tutorial.rst | 6 +++--- docs/news.rst | 44 ++++++++++++++++++++++++---------------- docs/topics/items.rst | 4 ++++ docs/topics/settings.rst | 2 ++ 4 files changed, 35 insertions(+), 21 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index a190ce407..0629b9e19 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -78,9 +78,9 @@ Our first Spider Spiders are classes that you define and that Scrapy uses to scrape information from a website (or a group of websites). They must subclass -:class:`scrapy.Spider` and define the initial requests to make, optionally how -to follow links in the pages, and how to parse the downloaded page content to -extract data. +:class:`~scrapy.spiders.Spider` and define the initial requests to make, +optionally how to follow links in the pages, and how to parse the downloaded +page content to extract data. This is the code for our first Spider. Save it in a file named ``quotes_spider.py`` under the ``tutorial/spiders`` directory in your project:: diff --git a/docs/news.rst b/docs/news.rst index aac750601..c107e907c 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -71,7 +71,7 @@ New features ~~~~~~~~~~~~ * A new scheduler priority queue, - :class:`scrapy.pqueues.DownloaderAwarePriorityQueue`, may be + ``scrapy.pqueues.DownloaderAwarePriorityQueue``, may be :ref:`enabled ` for a significant scheduling improvement on crawls targetting multiple web domains, at the cost of no :setting:`CONCURRENT_REQUESTS_PER_IP` support (:issue:`3520`) @@ -150,9 +150,9 @@ Bug fixes :setting:`AWS_REGION_NAME`, :setting:`AWS_USE_SSL`, :setting:`AWS_VERIFY` (:issue:`3625`) -* Fixed a memory leak in :class:`~scrapy.pipelines.media.MediaPipeline` - affecting, for example, non-200 responses and exceptions from custom - middlewares (:issue:`3813`) +* Fixed a memory leak in ``scrapy.pipelines.media.MediaPipeline`` affecting, + for example, non-200 responses and exceptions from custom middlewares + (:issue:`3813`) * Requests with private callbacks are now correctly unserialized from disk (:issue:`3790`) @@ -301,7 +301,7 @@ Deprecations * The ``queuelib.PriorityQueue`` value for the :setting:`SCHEDULER_PRIORITY_QUEUE` setting is deprecated. Use - :class:`scrapy.pqueues.ScrapyPriorityQueue` instead. + ``scrapy.pqueues.ScrapyPriorityQueue`` instead. * ``process_request`` callbacks passed to :class:`~scrapy.spiders.Rule` that do not accept two arguments are deprecated. @@ -551,7 +551,7 @@ Scrapy 1.5.2 (2019-01-22) *The fix is backward incompatible*, it enables telnet user-password authentication by default with a random generated password. If you can't - upgrade right away, please consider setting :setting:`TELNET_CONSOLE_PORT` + upgrade right away, please consider setting :setting:`TELNETCONSOLE_PORT` out of its default value. See :ref:`telnet console ` documentation for more info @@ -758,7 +758,9 @@ Enjoy! (Or read on for the rest of changes in this release.) Deprecations and Backward Incompatible Changes ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -- Default to ``canonicalize=False`` in :class:`scrapy.linkextractors.LinkExtractor` +- Default to ``canonicalize=False`` in + :class:`scrapy.linkextractors.LinkExtractor + ` (:issue:`2537`, fixes :issue:`1941` and :issue:`1982`): **warning, this is technically backward-incompatible** - Enable memusage extension by default (:issue:`2539`, fixes :issue:`2187`); @@ -794,10 +796,13 @@ New Features - New ``data:`` URI download handler (:issue:`2334`, fixes :issue:`2156`) - Log cache directory when HTTP Cache is used (:issue:`2611`, fixes :issue:`2604`) - Warn users when project contains duplicate spider names (fixes :issue:`2181`) -- :class:`CaselessDict` now accepts ``Mapping`` instances and not only dicts (:issue:`2646`) -- :ref:`Media downloads `, with :class:`FilesPipelines` - or :class:`ImagesPipelines`, can now optionally handle HTTP redirects - using the new :setting:`MEDIA_ALLOW_REDIRECTS` setting (:issue:`2616`, fixes :issue:`2004`) +- ``scrapy.utils.datatypes.CaselessDict`` now accepts ``Mapping`` instances and + not only dicts (:issue:`2646`) +- :ref:`Media downloads `, with + :class:`~scrapy.pipelines.files.FilesPipeline` or + :class:`~scrapy.pipelines.images.ImagesPipeline`, can now optionally handle + HTTP redirects using the new :setting:`MEDIA_ALLOW_REDIRECTS` setting + (:issue:`2616`, fixes :issue:`2004`) - Accept non-complete responses from websites using a new :setting:`DOWNLOAD_FAIL_ON_DATALOSS` setting (:issue:`2590`, fixes :issue:`2586`) - Optional pretty-printing of JSON and XML items via @@ -817,8 +822,8 @@ Bug fixes - LinkExtractor now strips leading and trailing whitespaces from attributes (:issue:`2547`, fixes :issue:`1614`) -- Properly handle whitespaces in action attribute in :class:`FormRequest` - (:issue:`2548`) +- Properly handle whitespaces in action attribute in + :class:`~scrapy.http.FormRequest` (:issue:`2548`) - Buffer CONNECT response bytes from proxy until all HTTP headers are received (:issue:`2495`, fixes :issue:`2491`) - FTP downloader now works on Python 3, provided you use Twisted>=17.1 @@ -851,7 +856,8 @@ Cleanups & Refactoring fixes :issue:`2560`) - Add omitted ``self`` arguments in default project middleware template (:issue:`2595`) - Remove redundant ``slot.add_request()`` call in ExecutionEngine (:issue:`2617`) -- Catch more specific ``os.error`` exception in :class:`FSFilesStore` (:issue:`2644`) +- Catch more specific ``os.error`` exception in + ``scrapy.pipelines.files.FSFilesStore`` (:issue:`2644`) - Change "localhost" test server certificate (:issue:`2720`) - Remove unused ``MEMUSAGE_REPORT`` setting (:issue:`2576`) @@ -868,7 +874,8 @@ Documentation (:issue:`2477`, fixes :issue:`2475`) - FAQ: rewrite note on Python 3 support on Windows (:issue:`2690`) - Rearrange selector sections (:issue:`2705`) -- Remove ``__nonzero__`` from :class:`SelectorList` docs (:issue:`2683`) +- Remove ``__nonzero__`` from :class:`~scrapy.selector.SelectorList` + docs (:issue:`2683`) - Mention how to disable request filtering in documentation of :setting:`DUPEFILTER_CLASS` setting (:issue:`2714`) - Add sphinx_rtd_theme to docs setup readme (:issue:`2668`) @@ -2327,7 +2334,7 @@ Scrapy 0.18.0 (released 2013-08-09) - Moved persistent (on disk) queues to a separate project (queuelib_) which scrapy now depends on - Add scrapy commands using external libraries (:issue:`260`) - Added ``--pdb`` option to ``scrapy`` command line tool -- Added :meth:`XPathSelector.remove_namespaces` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`. +- Added :meth:`XPathSelector.remove_namespaces ` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`. - Several improvements to spider contracts - New default middleware named MetaRefreshMiddldeware that handles meta-refresh html tag redirections, - MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62 @@ -2448,7 +2455,7 @@ Scrapy changes: - added options ``-o`` and ``-t`` to the :command:`runspider` command - documented :doc:`topics/autothrottle` and added to extensions installed by default. You still need to enable it with :setting:`AUTOTHROTTLE_ENABLED` - major Stats Collection refactoring: removed separation of global/per-spider stats, removed stats-related signals (``stats_spider_opened``, etc). Stats are much simpler now, backward compatibility is kept on the Stats Collector API and signals. -- added :meth:`~scrapy.contrib.spidermiddleware.SpiderMiddleware.process_start_requests` method to spider middlewares +- added :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_start_requests` method to spider middlewares - dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info. - dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info. - dropped Stats Collector singleton. Stats can now be accessed through the Crawler.stats attribute. See the stats collection documentation for more info. @@ -2609,7 +2616,8 @@ The numbers like #NNN reference tickets in the old issue tracker (Trac) which is New features and improvements ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ -- Passed item is now sent in the ``item`` argument of the :signal:`item_passed` (#273) +- Passed item is now sent in the ``item`` argument of the :signal:`item_passed + ` (#273) - Added verbose option to ``scrapy version`` command, useful for bug reports (#298) - HTTP cache now stored by default in the project data dir (#279) - Added project data storage directory (#276, #277) diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 260f5882c..fbc888f47 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -240,6 +240,10 @@ Item objects Items replicate the standard `dict API`_, including its constructor. The only additional attribute provided by Items is: + .. automethod:: copy + + .. automethod:: deepcopy + .. attribute:: fields A dictionary containing *all declared fields* for this Item, not only diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 75e0af63b..350c3c4b9 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -796,6 +796,7 @@ Default: ``True`` Whether or not to use passive mode when initiating FTP transfers. +.. reqmeta:: ftp_password .. setting:: FTP_PASSWORD FTP_PASSWORD @@ -814,6 +815,7 @@ in ``Request`` meta. .. _RFC 1635: https://tools.ietf.org/html/rfc1635 +.. reqmeta:: ftp_user .. setting:: FTP_USER FTP_USER From f52148143be614779ccfdb34cac995b16f8264ff Mon Sep 17 00:00:00 2001 From: akhter wahab Date: Mon, 7 Oct 2019 23:28:33 +0500 Subject: [PATCH 008/113] Add dmg, iso & apk to ignored other extensions --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index ebf3cd7d8..6049c312c 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -35,7 +35,7 @@ IGNORED_EXTENSIONS = [ 'odp', # other - 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', + 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', 'dmg', 'iso', 'apk' ] From 877ef4269e5683bf87832e816cf20a3cac352440 Mon Sep 17 00:00:00 2001 From: akhter wahab Date: Wed, 9 Oct 2019 16:03:44 +0500 Subject: [PATCH 009/113] Add .webm to ignored video extensions --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index 6049c312c..5f6df9c73 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -28,7 +28,7 @@ IGNORED_EXTENSIONS = [ # video '3gp', 'asf', 'asx', 'avi', 'mov', 'mp4', 'mpg', 'qt', 'rm', 'swf', 'wmv', - 'm4a', 'm4v', 'flv', + 'm4a', 'm4v', 'flv', 'webm', # office suites 'xls', 'xlsx', 'ppt', 'pptx', 'pps', 'doc', 'docx', 'odt', 'ods', 'odg', From a25a2d5ee4755046d736efa0a03898d9d18671f9 Mon Sep 17 00:00:00 2001 From: akhter wahab Date: Wed, 9 Oct 2019 16:05:39 +0500 Subject: [PATCH 010/113] Add .tar.xz to ignored other extensions --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index 5f6df9c73..0405462d6 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -35,7 +35,7 @@ IGNORED_EXTENSIONS = [ 'odp', # other - 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', 'dmg', 'iso', 'apk' + 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', 'dmg', 'iso', 'apk', 'tar.xz' ] From 532770df5226757498a941746cf94ae47043e101 Mon Sep 17 00:00:00 2001 From: akhter wahab Date: Wed, 9 Oct 2019 22:53:14 +0500 Subject: [PATCH 011/113] instead of .tar.xz adding .xz in others extensions --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index 0405462d6..3c75e683d 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -35,7 +35,7 @@ IGNORED_EXTENSIONS = [ 'odp', # other - 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', 'dmg', 'iso', 'apk', 'tar.xz' + 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', 'dmg', 'iso', 'apk', 'xz' ] From f0db1b4b6601e71caafe91f7970c9bfaa3d55395 Mon Sep 17 00:00:00 2001 From: "Matsievskiy S.V" Date: Fri, 11 Oct 2019 00:55:18 +0300 Subject: [PATCH 012/113] update zsh completion --- extras/scrapy_zsh_completion | 223 ++++++++++++++++++++++++++++++++--- 1 file changed, 204 insertions(+), 19 deletions(-) diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion index 564991aa8..86c52c36c 100644 --- a/extras/scrapy_zsh_completion +++ b/extras/scrapy_zsh_completion @@ -1,25 +1,210 @@ #compdef scrapy - -# zsh completion for the Scrapy command-line tool - _scrapy() { - local curcontext="$curcontext" cmd spiders + local context state state_descr line typeset -A opt_args - cmd=$words[2] - - case "$cmd" in - crawl|edit|check) - spiders=$(scrapy list 2>/dev/null) || spiders="" - if [[ -n "$spiders" ]]; then - compadd `echo $spiders` - fi - ;; - *) - if [[ CURRENT -eq 2 ]]; then - _arguments '*: :(check crawl edit fetch genspider list parse runspider settings shell startproject version view)' - fi - ;; + _arguments \ + "(- 1 *)--help[Help]" \ + "1: :->command" \ + "*:: :->args" + + case $state in + command) + _scrapy_cmds + ;; + args) + case $words[1] in + bench) + _scrapy_glb_opts + ;; + fetch) + local options=( + '--headers[print response HTTP headers instead of body]' + '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' + '--spider[use this spider]:spider:_scrapy_spiders' + '1::URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + genspider) + local options=( + {-l,--list}'[List available templates]' + {-e,--edit}'[Edit spider after creating it]' + '--force[If the spider already exists, overwrite it with the template]' + {-d,--dump=}'[Dump template to standard output]:template:(basic crawl csvfeed xmlfeed)' + {-t,--template=}'[Uses a custom template]:template:(basic crawl csvfeed xmlfeed)' + '1:name:(NAME)' + '2:domain:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + runspider) + local options=( + {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files' + {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)' + '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' + '1:spider file:_files -g \*.py' + ) + _scrapy_glb_opts $options + ;; + settings) + local options=( + '--get=[print raw setting value]:option:(SETTING)' + '--getbool=[print setting value, interpreted as a boolean]:option:(SETTING)' + '--getint=[print setting value, interpreted as an integer]:option:(SETTING)' + '--getfloat=[print setting value, interpreted as a float]:option:(SETTING)' + '--getlist=[print setting value, interpreted as a list]:option:(SETTING)' + ) + _scrapy_glb_opts $options + ;; + shell) + local options=( + '-c[evaluate the code in the shell, print the result and exit]:code:(CODE)' + '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' + '--spider[use this spider]:spider:_scrapy_spiders' + '::file:_files -g \*.http' + '::URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + startproject) + local options=( + '1:name:(NAME)' + '2:dir:_dir_list' + ) + _scrapy_glb_opts $options + ;; + version) + local options=( + {-v,--verbose}'[also display twisted/python/platform info (useful for bug reports)]' + ) + _scrapy_glb_opts $options + ;; + view) + local options=( + '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' + '--spider[use this spider]:spider:_scrapy_spiders' + '1:URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + check) + local options=( + '(- 1 *)'{-l,--list}'[only list contracts, without checking them]' + {-v,--verbose}'[print contract tests for all spiders]' + '1:spider:_scrapy_spiders' + ) + _scrapy_glb_opts $options + ;; + crawl) + local options=( + {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files' + {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)' + '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' + '1:spider:_scrapy_spiders' + ) + _scrapy_glb_opts $options + ;; + edit) + local options=( + '1:spider:_scrapy_spiders' + ) + _scrapy_glb_opts $options + ;; + list) + _scrapy_glb_opts + ;; + parse) + local options=( + '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' + '--spider[use this spider without looking for one]:spider:_scrapy_spiders' + '--pipelines[process items through pipelines]' + "--nolinks[don't show links to follow (extracted requests)]" + "--noitems[don't show scraped items]" + '--nocolour[avoid using pygments to colorize the output]' + {-r,--rules}'[use CrawlSpider rules to discover the callback]' + {-c,--callback=}'[use this callback for parsing, instead looking for a callback]:callback:(CALLBACK)' + {-m,--meta=}'[inject extra meta into the Request, it must be a valid raw json string]:meta:(META)' + '--cbkwargs=[inject extra callback kwargs into the Request, it must be a valid raw json string]:arguments:(CBKWARGS)' + {-d,--depth=}'[maximum depth for parsing requests (default: 1)]:depth:(DEPTH)' + {-v,--verbose}'[print each depth level one by one]' + '1:URL:_httpie_urls' + ) + _scrapy_glb_opts $options + ;; + esac + ;; esac } -_scrapy \ No newline at end of file +_scrapy_cmds() { + local -a commands project_commands + commands=( + 'bench:Run quick benchmark test' + 'fetch:Fetch a URL using the Scrapy downloader' + 'genspider:Generate new spider using pre-defined templates' + 'runspider:Run a self-contained spider (without creating a project)' + 'settings:Get settings values' + 'shell:Interactive scraping console' + 'startproject:Create new project' + 'version:Print Scrapy version' + 'view:Open URL in browser, as seen by Scrapy' + ) + project_commands=( + 'check:Check spider contracts' + 'crawl:Run a spider' + 'edit:Edit spider' + 'list:List available spiders' + 'parse:Parse URL (using its spider) and print the results' + ) + if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then + commands=(${commands[@]} ${project_commands[@]}) + fi + _describe -t common-commands 'common commands' commands +} + +_scrapy_glb_opts() { + local -a options + options=( + '(- *)'{-h,--help}'[show this help message and exit]' + '(--nolog)--logfile=[log file. if omitted stderr will be used]:file:_files' + '--pidfile=[write process ID to FILE]:file:_files' + '--profile=[write python cProfile stats to FILE]:file:_files' + '(--nolog)'{-L,--loglevel=}'[log level (default: INFO)]:log level:(DEBUG INFO WARN ERROR)' + '(-L --loglevel --logfile)--nolog[disable logging completely]' + '--pdb[enable pdb on failure]' + '*'{-s,--set=}'[set/override setting (may be repeated)]:value pair:(NAME=VALUE)' + ) + options=(${options[@]} "$@") + _arguments $options +} + +_httpie_urls() { + + local ret=1 + + if ! [[ -prefix [-+.a-z0-9]#:// ]]; then + local expl + compset -S '[^:/]*' && compstate[to_end]='' + _wanted url-schemas expl 'URL schema' compadd -S '' http:// https:// && ret=0 + else + _urls && ret=0 + fi + + return $ret + +} + +_scrapy_spiders() { + + local ret=1 + + if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then + compadd -S '' $(scrapy list) && ret=0 + else + compadd -S '' SPIDER && ret=0 + fi + + return $ret +} + +_scrapy $@ From 12f1e468e9f071e8f5c00921d0fbada460e41c57 Mon Sep 17 00:00:00 2001 From: Purva Udai Date: Sun, 13 Oct 2019 15:55:27 +0530 Subject: [PATCH 013/113] Issue #3731 --- scrapy/extensions/feedexport.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index ce2846eba..981efee55 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -199,9 +199,9 @@ class FeedExporter(object): def __init__(self, settings): self.settings = settings - self.urifmt = settings['FEED_URI'] - if not self.urifmt: + if not settings['FEED_URI']: raise NotConfigured + self.urifmt=str(settings['FEED_URI']) self.format = settings['FEED_FORMAT'].lower() self.export_encoding = settings['FEED_EXPORT_ENCODING'] self.storages = self._load_components('FEED_STORAGES') From 7b1e69dec4cb1f1c66e9df6f754cdd280cff7e71 Mon Sep 17 00:00:00 2001 From: Baron Hou Date: Tue, 15 Oct 2019 20:51:15 +0800 Subject: [PATCH 014/113] =?UTF-8?q?reponse=20=E2=86=92=20response=20(#4079?= =?UTF-8?q?)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/topics/developer-tools.rst | 2 +- docs/topics/downloader-middleware.rst | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index dcf8af365..bf14643be 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -203,7 +203,7 @@ where our quotes are coming from: First click on the request with the name ``scroll``. On the right you can now inspect the request. In ``Headers`` you'll find details about the request headers, such as the URL, the method, the IP-address, -and so on. We'll ignore the other tabs and click directly on ``Reponse``. +and so on. We'll ignore the other tabs and click directly on ``Response``. What you should see in the ``Preview`` pane is the rendered HTML-code, that is exactly what we saw when we called ``view(response)`` in the diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 539832618..2892b9b79 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -534,7 +534,7 @@ defines the methods described below. :param spider: the spider which generated the request :type spider: :class:`~scrapy.spiders.Spider` object - :param request: the request to find cached reponse for + :param request: the request to find cached response for :type request: :class:`~scrapy.http.Request` object .. method:: store_response(spider, request, response) From d72ed46fe01f95e77554e58588675d5a1ff47e0a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 15 Oct 2019 16:03:42 +0200 Subject: [PATCH 015/113] Improve how extra Item API members are introduced in the documentation --- docs/topics/items.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/topics/items.rst b/docs/topics/items.rst index fbc888f47..d70e7428b 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -237,8 +237,8 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard `dict API`_, including its constructor. The - only additional attribute provided by Items is: + Items replicate the standard `dict API`_, including its constructor, and + also provide the following additional API members: .. automethod:: copy From c9614a5bdd1c8a3f50eeace148c59f0b52e293a2 Mon Sep 17 00:00:00 2001 From: Bulat Date: Wed, 16 Oct 2019 12:07:19 +0300 Subject: [PATCH 016/113] Fixed BOT_NAME documentation --- docs/topics/settings.rst | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 75e0af63b..dbe3e9e44 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -229,8 +229,7 @@ BOT_NAME Default: ``'scrapybot'`` The name of the bot implemented by this Scrapy project (also known as the -project name). This will be used to construct the User-Agent by default, and -also for logging. +project name) and also for logging. It's automatically populated with your project name when you create your project with the :command:`startproject` command. From 84be6a941e3c4e80802dd52e43eb8e7064892cdb Mon Sep 17 00:00:00 2001 From: Bulat Date: Wed, 16 Oct 2019 14:04:07 +0300 Subject: [PATCH 017/113] Refactor sentence. --- docs/topics/settings.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index dbe3e9e44..d41381dd1 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -229,7 +229,7 @@ BOT_NAME Default: ``'scrapybot'`` The name of the bot implemented by this Scrapy project (also known as the -project name) and also for logging. +project name). This name will be used for the logging too. It's automatically populated with your project name when you create your project with the :command:`startproject` command. From 865d58fd1b5676dcca44b64970b5cd48fc1ab715 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jos=C3=A9=20Alberto=20Orejuela=20Garc=C3=ADa?= Date: Thu, 3 Oct 2019 18:06:19 +0200 Subject: [PATCH 018/113] Make punctuation consistent --- CODE_OF_CONDUCT.md | 2 +- README.rst | 18 +++++++++--------- 2 files changed, 10 insertions(+), 10 deletions(-) diff --git a/CODE_OF_CONDUCT.md b/CODE_OF_CONDUCT.md index d477168eb..d1cd3e517 100644 --- a/CODE_OF_CONDUCT.md +++ b/CODE_OF_CONDUCT.md @@ -68,7 +68,7 @@ members of the project's leadership. ## Attribution This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4, -available at [http://contributor-covenant.org/version/1/4][version] +available at [http://contributor-covenant.org/version/1/4][version]. [homepage]: http://contributor-covenant.org [version]: http://contributor-covenant.org/version/1/4/ diff --git a/README.rst b/README.rst index bd82bff06..87eaac2af 100644 --- a/README.rst +++ b/README.rst @@ -34,8 +34,8 @@ Scrapy is a fast high-level web crawling and web scraping framework, used to crawl websites and extract structured data from their pages. It can be used for a wide range of purposes, from data mining to monitoring and automated testing. -For more information including a list of features check the Scrapy homepage at: -https://scrapy.org +Check the Scrapy homepage at https://scrapy.org for more information, +including a list of features. Requirements ============ @@ -50,8 +50,8 @@ The quick way:: pip install scrapy -For more details see the install section in the documentation: -https://docs.scrapy.org/en/latest/intro/install.html +See the install section in the documentation at +https://docs.scrapy.org/en/latest/intro/install.html for more details. Documentation ============= @@ -62,17 +62,17 @@ directory. Releases ======== -You can find release notes at https://docs.scrapy.org/en/latest/news.html +You can check https://docs.scrapy.org/en/latest/news.html for release notes. Community (blog, twitter, mail list, IRC) ========================================= -See https://scrapy.org/community/ +See https://scrapy.org/community/ for details. Contributing ============ -See https://docs.scrapy.org/en/master/contributing.html +See https://docs.scrapy.org/en/master/contributing.html for details. Code of Conduct --------------- @@ -86,9 +86,9 @@ Please report unacceptable behavior to opensource@scrapinghub.com. Companies using Scrapy ====================== -See https://scrapy.org/companies/ +See https://scrapy.org/companies/ for a list. Commercial Support ================== -See https://scrapy.org/support/ +See https://scrapy.org/support/ for details. From 68a7d05ed86331ae70778c83a131e6dcc8b4ffc8 Mon Sep 17 00:00:00 2001 From: Ammar Najjar Date: Mon, 21 Oct 2019 15:42:24 +0200 Subject: [PATCH 019/113] docs: use __init__ method instead of constructor Issue #4086 --- docs/conf.py | 2 +- docs/news.rst | 10 ++++---- docs/topics/email.rst | 4 ++-- docs/topics/exporters.rst | 40 +++++++++++++++---------------- docs/topics/extensions.rst | 2 +- docs/topics/items.rst | 8 +++---- docs/topics/loaders.rst | 26 ++++++++++---------- docs/topics/request-response.rst | 14 +++++------ scrapy/exporters.py | 2 +- scrapy/extensions/feedexport.py | 2 +- scrapy/utils/datatypes.py | 2 +- scrapy/utils/misc.py | 6 ++--- scrapy/utils/python.py | 2 +- sep/sep-009.rst | 12 +++++----- tests/test_http_request.py | 6 ++--- tests/test_http_response.py | 2 +- tests/test_loader.py | 12 +++++----- tests/test_spider.py | 4 ++-- tests/test_utils_misc/__init__.py | 10 ++++---- 19 files changed, 83 insertions(+), 83 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 34dd5bcb7..6ab5959d5 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -237,7 +237,7 @@ coverage_ignore_pyobjects = [ r'\bContractsManager\b$', # For default contracts we only want to document their general purpose in - # their constructor, the methods they reimplement to achieve that purpose + # their __init__ method, the methods they reimplement to achieve that purpose # should be irrelevant to developers using those contracts. r'\w+Contract\.(adjust_request_args|(pre|post)_process)$', diff --git a/docs/news.rst b/docs/news.rst index 59317f5eb..c1f548060 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -84,12 +84,12 @@ New features convenient way to build JSON requests (:issue:`3504`, :issue:`3505`) * A ``process_request`` callback passed to the :class:`~scrapy.spiders.Rule` - constructor now receives the :class:`~scrapy.http.Response` object that + ``__init__`` method now receives the :class:`~scrapy.http.Response` object that originated the request as its second argument (:issue:`3682`) * A new ``restrict_text`` parameter for the :attr:`LinkExtractor ` - constructor allows filtering links by linking text (:issue:`3622`, + ``__init__`` method allows filtering links by linking text (:issue:`3622`, :issue:`3635`) * A new :setting:`FEED_STORAGE_S3_ACL` setting allows defining a custom ACL @@ -255,7 +255,7 @@ The following deprecated APIs have been removed (:issue:`3578`): * From :class:`~scrapy.selector.Selector`: - * ``_root`` (both the constructor argument and the object property, use + * ``_root`` (both the ``__init__`` method argument and the object property, use ``root``) * ``extract_unquoted`` (use ``getall``) @@ -2479,7 +2479,7 @@ Scrapy changes: - removed ``ENCODING_ALIASES`` setting, as encoding auto-detection has been moved to the `w3lib`_ library - promoted :ref:`topics-djangoitem` to main contrib - LogFormatter method now return dicts(instead of strings) to support lazy formatting (:issue:`164`, :commit:`dcef7b0`) -- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the constructor +- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the ``__init__`` method - replaced memory usage acounting with (more portable) `resource`_ module, removed ``scrapy.utils.memory`` module - removed signal: ``scrapy.mail.mail_sent`` - removed ``TRACK_REFS`` setting, now :ref:`trackrefs ` is always enabled @@ -2693,7 +2693,7 @@ API changes - ``Request.copy()`` and ``Request.replace()`` now also copies their ``callback`` and ``errback`` attributes (#231) - Removed ``UrlFilterMiddleware`` from ``scrapy.contrib`` (already disabled by default) - Offsite middelware doesn't filter out any request coming from a spider that doesn't have a allowed_domains attribute (#225) -- Removed Spider Manager ``load()`` method. Now spiders are loaded in the constructor itself. +- Removed Spider Manager ``load()`` method. Now spiders are loaded in the ``__init__`` method itself. - Changes to Scrapy Manager (now called "Crawler"): - ``scrapy.core.manager.ScrapyManager`` class renamed to ``scrapy.crawler.Crawler`` - ``scrapy.core.manager.scrapymanager`` singleton moved to ``scrapy.project.crawler`` diff --git a/docs/topics/email.rst b/docs/topics/email.rst index 949cdc638..12eedf2cd 100644 --- a/docs/topics/email.rst +++ b/docs/topics/email.rst @@ -21,7 +21,7 @@ Quick example ============= There are two ways to instantiate the mail sender. You can instantiate it using -the standard constructor:: +the standard ``__init__`` method:: from scrapy.mail import MailSender mailer = MailSender() @@ -111,7 +111,7 @@ uses `Twisted non-blocking IO`_, like the rest of the framework. Mail settings ============= -These settings define the default constructor values of the :class:`MailSender` +These settings define the default ``__init__`` method values of the :class:`MailSender` class, and can be used to configure e-mail notifications in your project without writing any code (for those extensions and code that uses :class:`MailSender`). diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index a698a6a4e..d7ab7a5cc 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -87,8 +87,8 @@ described next. 1. Declaring a serializer in the field -------------------------------------- -If you use :class:`~.Item` you can declare a serializer in the -:ref:`field metadata `. The serializer must be +If you use :class:`~.Item` you can declare a serializer in the +:ref:`field metadata `. The serializer must be a callable which receives a value and returns its serialized form. Example:: @@ -144,7 +144,7 @@ BaseItemExporter defining what fields to export, whether to export empty fields, or which encoding to use. - These features can be configured through the constructor arguments which + These features can be configured through the `__init__` method arguments which populate their respective instance attributes: :attr:`fields_to_export`, :attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`. @@ -246,8 +246,8 @@ XmlItemExporter :param item_element: The name of each item element in the exported XML. :type item_element: str - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor. + The additional keyword arguments of this `__init__` method are passed to the + :class:`BaseItemExporter` `__init__` method A typical output of this exporter would be:: @@ -306,9 +306,9 @@ CsvItemExporter multi-valued fields, if found. :type include_headers_line: str - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor, and the leftover arguments to the - `csv.writer`_ constructor, so you can use any ``csv.writer`` constructor + The additional keyword arguments of this `__init__` method are passed to the + :class:`BaseItemExporter` `__init__` method, and the leftover arguments to the + `csv.writer`_ `__init__` method, so you can use any ``csv.writer`` `__init__` method argument to customize this exporter. A typical output of this exporter would be:: @@ -334,8 +334,8 @@ PickleItemExporter For more information, refer to the `pickle module documentation`_. - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor. + The additional keyword arguments of this `__init__` method are passed to the + :class:`BaseItemExporter` `__init__` method. Pickle isn't a human readable format, so no output examples are provided. @@ -351,8 +351,8 @@ PprintItemExporter :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) - The additional keyword arguments of this constructor are passed to the - :class:`BaseItemExporter` constructor. + The additional keyword arguments of this `__init__` method are passed to the + :class:`BaseItemExporter` `__init__` method A typical output of this exporter would be:: @@ -367,10 +367,10 @@ JsonItemExporter .. class:: JsonItemExporter(file, \**kwargs) Exports Items in JSON format to the specified file-like object, writing all - objects as a list of objects. The additional constructor arguments are - passed to the :class:`BaseItemExporter` constructor, and the leftover - arguments to the `JSONEncoder`_ constructor, so you can use any - `JSONEncoder`_ constructor argument to customize this exporter. + objects as a list of objects. The additional `__init__` method arguments are + passed to the :class:`BaseItemExporter` `__init__` method, and the leftover + arguments to the `JSONEncoder`_ `__init__` method, so you can use any + `JSONEncoder`_ `__init__` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) @@ -398,10 +398,10 @@ JsonLinesItemExporter .. class:: JsonLinesItemExporter(file, \**kwargs) Exports Items in JSON format to the specified file-like object, writing one - JSON-encoded item per line. The additional constructor arguments are passed - to the :class:`BaseItemExporter` constructor, and the leftover arguments to - the `JSONEncoder`_ constructor, so you can use any `JSONEncoder`_ - constructor argument to customize this exporter. + JSON-encoded item per line. The additional `__init__` method arguments are passed + to the :class:`BaseItemExporter` `__init__` method and the leftover arguments to + the `JSONEncoder`_ `__init__` method, so you can use any `JSONEncoder`_ + `__init__` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index 72c2290b5..0a7455ec9 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -28,7 +28,7 @@ Loading & activating extensions Extensions are loaded and activated at startup by instantiating a single instance of the extension class. Therefore, all the extension initialization -code must be performed in the class constructor (``__init__`` method). +code must be performed in the class ``__init__`` method. To make an extension available, add it to the :setting:`EXTENSIONS` setting in your Scrapy settings. In :setting:`EXTENSIONS`, each extension is represented diff --git a/docs/topics/items.rst b/docs/topics/items.rst index d70e7428b..8ea2635cc 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -16,12 +16,12 @@ especially in a larger project with many spiders. To define common output data format Scrapy provides the :class:`Item` class. :class:`Item` objects are simple containers used to collect the scraped data. They provide a `dictionary-like`_ API with a convenient syntax for declaring -their available fields. +their available fields. -Various Scrapy components use extra information provided by Items: +Various Scrapy components use extra information provided by Items: exporters look at declared fields to figure out columns to export, serialization can be customized using Item fields metadata, :mod:`trackref` -tracks Item instances to help find memory leaks +tracks Item instances to help find memory leaks (see :ref:`topics-leaks-trackrefs`), etc. .. _dictionary-like: https://docs.python.org/2/library/stdtypes.html#dict @@ -237,7 +237,7 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard `dict API`_, including its constructor, and + Items replicate the standard `dict API`_, including its `__init__` method and also provide the following additional API members: .. automethod:: copy diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 1c2f1da4d..db1175a13 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -26,7 +26,7 @@ Using Item Loaders to populate items To use an Item Loader, you must first instantiate it. You can either instantiate it with a dict-like object (e.g. Item or dict) or without one, in -which case an Item is automatically instantiated in the Item Loader constructor +which case an Item is automatically instantiated in the Item Loader ``__init__`` method using the Item class specified in the :attr:`ItemLoader.default_item_class` attribute. @@ -265,7 +265,7 @@ There are several ways to modify Item Loader context values: loader.context['unit'] = 'cm' 2. On Item Loader instantiation (the keyword arguments of Item Loader - constructor are stored in the Item Loader context):: + ``__init__`` methodare stored in the Item Loader context):: loader = ItemLoader(product, unit='cm') @@ -494,7 +494,7 @@ ItemLoader objects .. attribute:: default_item_class An Item class (or factory), used to instantiate items when not given in - the constructor. + the `__init__` method .. attribute:: default_input_processor @@ -509,15 +509,15 @@ ItemLoader objects .. attribute:: default_selector_class The class used to construct the :attr:`selector` of this - :class:`ItemLoader`, if only a response is given in the constructor. - If a selector is given in the constructor this attribute is ignored. + :class:`ItemLoader`, if only a response is given in the `__init__` method + If a selector is given in the `__init__` method this attribute is ignored. This attribute is sometimes overridden in subclasses. .. attribute:: selector The :class:`~scrapy.selector.Selector` object to extract data from. - It's either the selector given in the constructor or one created from - the response given in the constructor using the + It's either the selector given in the `__init__` methodor one created from + the response given in the `__init__` methodusing the :attr:`default_selector_class`. This attribute is meant to be read-only. @@ -642,7 +642,7 @@ Here is a list of all built-in processors: .. class:: Identity The simplest processor, which doesn't do anything. It returns the original - values unchanged. It doesn't receive any constructor arguments, nor does it + values unchanged. It doesn't receive any `__init__` method arguments, nor does it accept Loader contexts. Example:: @@ -656,7 +656,7 @@ Here is a list of all built-in processors: Returns the first non-null/non-empty value from the values received, so it's typically used as an output processor to single-valued fields. - It doesn't receive any constructor arguments, nor does it accept Loader contexts. + It doesn't receive any `__init__` methodarguments, nor does it accept Loader contexts. Example:: @@ -667,7 +667,7 @@ Here is a list of all built-in processors: .. class:: Join(separator=u' ') - Returns the values joined with the separator given in the constructor, which + Returns the values joined with the separator given in the `__init__` method which defaults to ``u' '``. It doesn't accept Loader contexts. When using the default separator, this processor is equivalent to the @@ -705,7 +705,7 @@ Here is a list of all built-in processors: those which do, this processor will pass the currently active :ref:`Loader context ` through that parameter. - The keyword arguments passed in the constructor are used as the default + The keyword arguments passed in the `__init__` methodare used as the default Loader context values passed to each function call. However, the final Loader context values passed to functions are overridden with the currently active Loader context accessible through the :meth:`ItemLoader.context` @@ -749,12 +749,12 @@ Here is a list of all built-in processors: ['HELLO, 'THIS', 'IS', 'SCRAPY'] As with the Compose processor, functions can receive Loader contexts, and - constructor keyword arguments are used as default context values. See + `__init__` method keyword arguments are used as default context values. See :class:`Compose` processor for more info. .. class:: SelectJmes(json_path) - Queries the value using the json path provided to the constructor and returns the output. + Queries the value using the json path provided to the `__init__` methodand returns the output. Requires jmespath (https://github.com/jmespath/jmespath.py) to run. This processor takes only one input at a time. diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 727c67482..d253064f2 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -137,7 +137,7 @@ Request objects A string containing the URL of this request. Keep in mind that this attribute contains the escaped URL, so it can differ from the URL passed in - the constructor. + the `__init__` method This attribute is read-only. To change the URL of a Request use :meth:`replace`. @@ -400,7 +400,7 @@ fields with form data from :class:`Response` objects. .. class:: FormRequest(url, [formdata, ...]) - The :class:`FormRequest` class adds a new keyword parameter to the constructor. The + The :class:`FormRequest` class adds a new keyword parameter to the `__init__` method The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -473,7 +473,7 @@ fields with form data from :class:`Response` objects. :type dont_click: boolean The other parameters of this class method are passed directly to the - :class:`FormRequest` constructor. + :class:`FormRequest` `__init__` method .. versionadded:: 0.10.3 The ``formname`` parameter. @@ -547,7 +547,7 @@ dealing with JSON requests. .. class:: JsonRequest(url, [... data, dumps_kwargs]) - The :class:`JsonRequest` class adds two new keyword parameters to the constructor. The + The :class:`JsonRequest` class adds two new keyword parameters to the `__init__` method The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -556,7 +556,7 @@ dealing with JSON requests. :param data: is any JSON serializable object that needs to be JSON encoded and assigned to body. if :attr:`Request.body` argument is provided this parameter will be ignored. - if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be + if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be set to ``'POST'`` automatically. :type data: JSON serializable object @@ -721,7 +721,7 @@ TextResponse objects :class:`Response` class, which is meant to be used only for binary data, such as images, sounds or any media file. - :class:`TextResponse` objects support a new constructor argument, in + :class:`TextResponse` objects support a new `__init__` method argument, in addition to the base :class:`Response` objects. The remaining functionality is the same as for the :class:`Response` class and is not documented here. @@ -755,7 +755,7 @@ TextResponse objects A string with the encoding of this response. The encoding is resolved by trying the following mechanisms, in order: - 1. the encoding passed in the constructor ``encoding`` argument + 1. the encoding passed in the `__init__` method`` ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this encoding is not valid (ie. unknown), it is ignored and the next diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 6fc87ed18..2fdc86b63 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -31,7 +31,7 @@ class BaseItemExporter(object): def _configure(self, options, dont_fail=False): """Configure the exporter by poping options from the ``options`` dict. If dont_fail is set, it won't raise an exception on unexpected options - (useful for using with keyword arguments in subclasses constructors) + (useful for using with keyword arguments in subclasses __init__ methods) """ self.encoding = options.pop('encoding', None) self.fields_to_export = options.pop('fields_to_export', None) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index ce2846eba..655c10482 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -106,7 +106,7 @@ class S3FeedStorage(BlockingFeedStorage): warnings.warn( "Initialising `scrapy.extensions.feedexport.S3FeedStorage` " "without AWS keys is deprecated. Please supply credentials or " - "use the `from_crawler()` constructor.", + "use the `from_crawler()` __init__ method.", category=ScrapyDeprecationWarning, stacklevel=2 ) diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index df2b99c28..0dfee1b27 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -246,7 +246,7 @@ class CaselessDict(dict): class MergeDict(object): """ A simple class for creating new "virtual" dictionaries that actually look - up values in more than one dictionary, passed in the constructor. + up values in more than one dictionary, passed in the __init__ method. If a key appears in more than one of the given dictionaries, only the first occurrence will be used. diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index f638adb25..c892bd438 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -123,14 +123,14 @@ def rel_has_nofollow(rel): def create_instance(objcls, settings, crawler, *args, **kwargs): """Construct a class instance using its ``from_crawler`` or - ``from_settings`` constructors, if available. + ``from_settings`` __init__ method, if available. At least one of ``settings`` and ``crawler`` needs to be different from ``None``. If ``settings `` is ``None``, ``crawler.settings`` will be used. - If ``crawler`` is ``None``, only the ``from_settings`` constructor will be + If ``crawler`` is ``None``, only the ``from_settings`` __init__ method will be tried. - ``*args`` and ``**kwargs`` are forwarded to the constructors. + ``*args`` and ``**kwargs`` are forwarded to the __init__ methods. Raises ``ValueError`` if both ``settings`` and ``crawler`` are ``None``. """ diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index c6140f885..0d645543a 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -303,7 +303,7 @@ class WeakKeyCache(object): def stringify_dict(dct_or_tuples, encoding='utf-8', keys_only=True): """Return a (new) dict with unicode keys (and values when "keys_only" is False) of the given dict converted to strings. ``dct_or_tuples`` can be a - dict or a list of tuples, like any dict constructor supports. + dict or a list of tuples, like any dict __init__ method supports. """ d = {} for k, v in six.iteritems(dict(dct_or_tuples)): diff --git a/sep/sep-009.rst b/sep/sep-009.rst index 232a536a8..929cedc61 100644 --- a/sep/sep-009.rst +++ b/sep/sep-009.rst @@ -38,7 +38,7 @@ singletons members of that object, as explained below: ``scrapy.core.manager.ExecutionManager``) - instantiated with a ``Settings`` object - - **crawler.settings**: ``scrapy.conf.Settings`` instance (passed in the constructor) + - **crawler.settings**: ``scrapy.conf.Settings`` instance (passed in the ``__init__`` method) - **crawler.extensions**: ``scrapy.extension.ExtensionManager`` instance - **crawler.engine**: ``scrapy.core.engine.ExecutionEngine`` instance - ``crawler.engine.scheduler`` @@ -55,7 +55,7 @@ singletons members of that object, as explained below: ``STATS_CLASS`` setting) - **crawler.log**: Logger class with methods replacing the current ``scrapy.log`` functions. Logging would be started (if enabled) on - ``Crawler`` constructor, so no log starting functions are required. + ``Crawler`` __init__ method, so no log starting functions are required. - ``crawler.log.msg`` - **crawler.signals**: signal handling @@ -69,12 +69,12 @@ Required code changes after singletons removal ============================================== All components (extensions, middlewares, etc) will receive this ``Crawler`` -object in their constructors, and this will be the only mechanism for accessing +object in their ``__init__`` methods, and this will be the only mechanism for accessing any other components (as opposed to importing each singleton from their respective module). This will also serve to stabilize the core API, something which we haven't documented so far (partly because of this). -So, for a typical middleware constructor code, instead of this: +So, for a typical middleware ``__init__`` method code, instead of this: :: @@ -125,13 +125,13 @@ Open issues to resolve - Should we pass ``Settings`` object to ``ScrapyCommand.add_options()``? - How should spiders access settings? - - Option 1. Pass ``Crawler`` object to spider constructors too + - Option 1. Pass ``Crawler`` object to spider ``__init__`` methods too - pro: one way to access all components (settings and signals being the most relevant to spiders) - con?: spider code can access (and control) any crawler component - since we don't want to support spiders messing with the crawler (write an extension or spider middleware if you need that) - - Option 2. Pass ``Settings`` object to spider constructors, which would + - Option 2. Pass ``Settings`` object to spider ``__init__`` methods, which would then be accessed through ``self.settings``, like logging which is accessed through ``self.log`` diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 16d7a1cb8..96a4fb141 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -25,7 +25,7 @@ class RequestTest(unittest.TestCase): default_meta = {} def test_init(self): - # Request requires url in the constructor + # Request requires url in the __init__ method self.assertRaises(Exception, self.request_class) # url argument must be basestring @@ -500,7 +500,7 @@ class FormRequestTest(RequestTest): formdata=(('foo', 'bar'), ('foo', 'baz'))) self.assertEqual(urlparse(req.url).hostname, 'www.example.com') self.assertEqual(urlparse(req.url).query, 'foo=bar&foo=baz') - + def test_from_response_override_duplicate_form_key(self): response = _buildresponse( """
@@ -657,7 +657,7 @@ class FormRequestTest(RequestTest): req = self.request_class.from_response(response, dont_click=True) fs = _qs(req) self.assertEqual(fs, {b'i1': [b'i1v'], b'i2': [b'i2v']}) - + def test_from_response_clickdata_does_not_ignore_image(self): response = _buildresponse( """ diff --git a/tests/test_http_response.py b/tests/test_http_response.py index cd5c3486e..dfc8562f3 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -534,7 +534,7 @@ class XmlResponseTest(TextResponseTest): r2 = self.response_class("http://www.example.com", body=body) self._assert_response_values(r2, 'iso-8859-1', body) - # make sure replace() preserves the explicit encoding passed in the constructor + # make sure replace() preserves the explicit encoding passed in the __init__ method body = b"""""" r3 = self.response_class("http://www.example.com", body=body, encoding='utf-8') body2 = b"New body" diff --git a/tests/test_loader.py b/tests/test_loader.py index 2725b001a..bcc1e6421 100644 --- a/tests/test_loader.py +++ b/tests/test_loader.py @@ -548,11 +548,11 @@ class SelectortemLoaderTest(unittest.TestCase): """) - def test_constructor(self): + def test_init_method(self): l = TestItemLoader() self.assertEqual(l.selector, None) - def test_constructor_errors(self): + def test_init_method_errors(self): l = TestItemLoader() self.assertRaises(RuntimeError, l.add_xpath, 'url', '//a/@href') self.assertRaises(RuntimeError, l.replace_xpath, 'url', '//a/@href') @@ -561,7 +561,7 @@ class SelectortemLoaderTest(unittest.TestCase): self.assertRaises(RuntimeError, l.replace_css, 'name', '#name::text') self.assertRaises(RuntimeError, l.get_css, '#name::text') - def test_constructor_with_selector(self): + def test_init_method_with_selector(self): sel = Selector(text=u"
marta
") l = TestItemLoader(selector=sel) self.assertIs(l.selector, sel) @@ -569,7 +569,7 @@ class SelectortemLoaderTest(unittest.TestCase): l.add_xpath('name', '//div/text()') self.assertEqual(l.get_output_value('name'), [u'Marta']) - def test_constructor_with_selector_css(self): + def test_init_method_with_selector_css(self): sel = Selector(text=u"
marta
") l = TestItemLoader(selector=sel) self.assertIs(l.selector, sel) @@ -577,14 +577,14 @@ class SelectortemLoaderTest(unittest.TestCase): l.add_css('name', 'div::text') self.assertEqual(l.get_output_value('name'), [u'Marta']) - def test_constructor_with_response(self): + def test_init_method_with_response(self): l = TestItemLoader(response=self.response) self.assertTrue(l.selector) l.add_xpath('name', '//div/text()') self.assertEqual(l.get_output_value('name'), [u'Marta']) - def test_constructor_with_response_css(self): + def test_init_method_with_response_css(self): l = TestItemLoader(response=self.response) self.assertTrue(l.selector) diff --git a/tests/test_spider.py b/tests/test_spider.py index 2220b8ffc..64ff40b61 100644 --- a/tests/test_spider.py +++ b/tests/test_spider.py @@ -42,12 +42,12 @@ class SpiderTest(unittest.TestCase): self.assertEqual(list(start_requests), []) def test_spider_args(self): - """Constructor arguments are assigned to spider attributes""" + """__init__ method arguments are assigned to spider attributes""" spider = self.spider_class('example.com', foo='bar') self.assertEqual(spider.foo, 'bar') def test_spider_without_name(self): - """Constructor arguments are assigned to spider attributes""" + """__init__ method arguments are assigned to spider attributes""" self.assertRaises(ValueError, self.spider_class) self.assertRaises(ValueError, self.spider_class, somearg='foo') diff --git a/tests/test_utils_misc/__init__.py b/tests/test_utils_misc/__init__.py index e109d5343..457a2aa78 100644 --- a/tests/test_utils_misc/__init__.py +++ b/tests/test_utils_misc/__init__.py @@ -109,11 +109,11 @@ class UtilsMiscTestCase(unittest.TestCase): else: mock.assert_called_once_with(*args, **kwargs) - # Check usage of correct constructor using four mocks: - # 1. with no alternative constructors - # 2. with from_settings() constructor - # 3. with from_crawler() constructor - # 4. with from_settings() and from_crawler() constructor + # Check usage of correct __init__ method using four mocks: + # 1. with no alternative __init__ methods + # 2. with from_settings() __init__ method + # 3. with from_crawler() __init__ method + # 4. with from_settings() and from_crawler() __init__ method spec_sets = ([], ['from_settings'], ['from_crawler'], ['from_settings', 'from_crawler']) for specs in spec_sets: From 3d4317bfe4697955de1c2450ce4cb0d12471a9a3 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 21 Oct 2019 18:32:30 +0100 Subject: [PATCH 020/113] [tox.ini] Added python 3.8 fields https://github.com/scrapy/scrapy/issues/4085 --- tox.ini | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/tox.ini b/tox.ini index ffe7360d3..14afec23f 100644 --- a/tox.ini +++ b/tox.ini @@ -92,6 +92,10 @@ deps = {[testenv:py35]deps} basepython = python3.7 deps = {[testenv:py35]deps} +[testenv:py38] +basepython = python3.8 +deps = {[testenv:py35]deps} + [testenv:pypy3] basepython = pypy3 deps = {[testenv:py35]deps} @@ -128,6 +132,13 @@ deps = reppy robotexclusionrulesparser +[testenv:py38-extra-deps] +basepython = python3.8 +deps = + {[testenv:py35]deps} + reppy + robotexclusionrulesparser + [testenv:py27-extra-deps] basepython = python2.7 deps = From 4e939ca75d14daf5bc658fdbe5a97af6f4c3c498 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 21 Oct 2019 18:33:18 +0100 Subject: [PATCH 021/113] [setup.py] Added python 3.8 fields https://github.com/scrapy/scrapy/issues/4085 --- setup.py | 1 + 1 file changed, 1 insertion(+) diff --git a/setup.py b/setup.py index 850456503..2f5fca4c9 100644 --- a/setup.py +++ b/setup.py @@ -56,6 +56,7 @@ setup( 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', + 'Programming Language :: Python :: 3.8', 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', From c12a075164e95e98a136499faf7bf8eefdce0300 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 21 Oct 2019 18:34:15 +0100 Subject: [PATCH 022/113] [.travis.yml] Added python 3.8 fields https://github.com/scrapy/scrapy/issues/4085 --- .travis.yml | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/.travis.yml b/.travis.yml index 0190a7f4d..2ba504972 100644 --- a/.travis.yml +++ b/.travis.yml @@ -27,6 +27,10 @@ matrix: python: 3.7 - env: TOXENV=py37-extra-deps python: 3.7 + - env: TOXENV=py38 + python: 3.8 + - env: TOXENV=py37-extra-deps + python: 3.8 - env: TOXENV=docs python: 3.6 install: From da8cd9448de28bdeac7343bcd367e95237423789 Mon Sep 17 00:00:00 2001 From: Ammar Najjar Date: Mon, 21 Oct 2019 19:48:13 +0200 Subject: [PATCH 023/113] docs: always surround __init__ with `` in docs Issue #4086 --- docs/topics/exporters.rst | 36 ++++++++++++++++---------------- docs/topics/items.rst | 2 +- docs/topics/loaders.rst | 22 +++++++++---------- docs/topics/request-response.rst | 12 +++++------ scrapy/exporters.py | 2 +- scrapy/extensions/feedexport.py | 2 +- scrapy/utils/datatypes.py | 2 +- scrapy/utils/misc.py | 6 +++--- scrapy/utils/python.py | 2 +- sep/sep-009.rst | 2 +- tests/test_spider.py | 4 ++-- 11 files changed, 46 insertions(+), 46 deletions(-) diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index d7ab7a5cc..1b8a69ca3 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -144,7 +144,7 @@ BaseItemExporter defining what fields to export, whether to export empty fields, or which encoding to use. - These features can be configured through the `__init__` method arguments which + These features can be configured through the ``__init__`` method arguments which populate their respective instance attributes: :attr:`fields_to_export`, :attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`. @@ -246,8 +246,8 @@ XmlItemExporter :param item_element: The name of each item element in the exported XML. :type item_element: str - The additional keyword arguments of this `__init__` method are passed to the - :class:`BaseItemExporter` `__init__` method + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method A typical output of this exporter would be:: @@ -306,9 +306,9 @@ CsvItemExporter multi-valued fields, if found. :type include_headers_line: str - The additional keyword arguments of this `__init__` method are passed to the - :class:`BaseItemExporter` `__init__` method, and the leftover arguments to the - `csv.writer`_ `__init__` method, so you can use any ``csv.writer`` `__init__` method + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the + `csv.writer`_ ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method argument to customize this exporter. A typical output of this exporter would be:: @@ -334,8 +334,8 @@ PickleItemExporter For more information, refer to the `pickle module documentation`_. - The additional keyword arguments of this `__init__` method are passed to the - :class:`BaseItemExporter` `__init__` method. + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method. Pickle isn't a human readable format, so no output examples are provided. @@ -351,8 +351,8 @@ PprintItemExporter :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) - The additional keyword arguments of this `__init__` method are passed to the - :class:`BaseItemExporter` `__init__` method + The additional keyword arguments of this ``__init__`` method are passed to the + :class:`BaseItemExporter` ``__init__`` method A typical output of this exporter would be:: @@ -367,10 +367,10 @@ JsonItemExporter .. class:: JsonItemExporter(file, \**kwargs) Exports Items in JSON format to the specified file-like object, writing all - objects as a list of objects. The additional `__init__` method arguments are - passed to the :class:`BaseItemExporter` `__init__` method, and the leftover - arguments to the `JSONEncoder`_ `__init__` method, so you can use any - `JSONEncoder`_ `__init__` method argument to customize this exporter. + objects as a list of objects. The additional ``__init__`` method arguments are + passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover + arguments to the `JSONEncoder`_ ``__init__`` method, so you can use any + `JSONEncoder`_ ``__init__`` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) @@ -398,10 +398,10 @@ JsonLinesItemExporter .. class:: JsonLinesItemExporter(file, \**kwargs) Exports Items in JSON format to the specified file-like object, writing one - JSON-encoded item per line. The additional `__init__` method arguments are passed - to the :class:`BaseItemExporter` `__init__` method and the leftover arguments to - the `JSONEncoder`_ `__init__` method, so you can use any `JSONEncoder`_ - `__init__` method argument to customize this exporter. + JSON-encoded item per line. The additional ``__init__`` method arguments are passed + to the :class:`BaseItemExporter` ``__init__`` method and the leftover arguments to + the `JSONEncoder`_ ``__init__`` method, so you can use any `JSONEncoder`_ + ``__init__`` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 8ea2635cc..370409026 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -237,7 +237,7 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard `dict API`_, including its `__init__` method and + Items replicate the standard `dict API`_, including its ``__init__`` method and also provide the following additional API members: .. automethod:: copy diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index db1175a13..72610f645 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -494,7 +494,7 @@ ItemLoader objects .. attribute:: default_item_class An Item class (or factory), used to instantiate items when not given in - the `__init__` method + the ``__init__`` method .. attribute:: default_input_processor @@ -509,15 +509,15 @@ ItemLoader objects .. attribute:: default_selector_class The class used to construct the :attr:`selector` of this - :class:`ItemLoader`, if only a response is given in the `__init__` method - If a selector is given in the `__init__` method this attribute is ignored. + :class:`ItemLoader`, if only a response is given in the ``__init__`` method + If a selector is given in the ``__init__`` method this attribute is ignored. This attribute is sometimes overridden in subclasses. .. attribute:: selector The :class:`~scrapy.selector.Selector` object to extract data from. - It's either the selector given in the `__init__` methodor one created from - the response given in the `__init__` methodusing the + It's either the selector given in the ``__init__`` methodor one created from + the response given in the ``__init__`` methodusing the :attr:`default_selector_class`. This attribute is meant to be read-only. @@ -642,7 +642,7 @@ Here is a list of all built-in processors: .. class:: Identity The simplest processor, which doesn't do anything. It returns the original - values unchanged. It doesn't receive any `__init__` method arguments, nor does it + values unchanged. It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts. Example:: @@ -656,7 +656,7 @@ Here is a list of all built-in processors: Returns the first non-null/non-empty value from the values received, so it's typically used as an output processor to single-valued fields. - It doesn't receive any `__init__` methodarguments, nor does it accept Loader contexts. + It doesn't receive any ``__init__`` methodarguments, nor does it accept Loader contexts. Example:: @@ -667,7 +667,7 @@ Here is a list of all built-in processors: .. class:: Join(separator=u' ') - Returns the values joined with the separator given in the `__init__` method which + Returns the values joined with the separator given in the ``__init__`` method which defaults to ``u' '``. It doesn't accept Loader contexts. When using the default separator, this processor is equivalent to the @@ -705,7 +705,7 @@ Here is a list of all built-in processors: those which do, this processor will pass the currently active :ref:`Loader context ` through that parameter. - The keyword arguments passed in the `__init__` methodare used as the default + The keyword arguments passed in the ``__init__`` methodare used as the default Loader context values passed to each function call. However, the final Loader context values passed to functions are overridden with the currently active Loader context accessible through the :meth:`ItemLoader.context` @@ -749,12 +749,12 @@ Here is a list of all built-in processors: ['HELLO, 'THIS', 'IS', 'SCRAPY'] As with the Compose processor, functions can receive Loader contexts, and - `__init__` method keyword arguments are used as default context values. See + ``__init__`` method keyword arguments are used as default context values. See :class:`Compose` processor for more info. .. class:: SelectJmes(json_path) - Queries the value using the json path provided to the `__init__` methodand returns the output. + Queries the value using the json path provided to the ``__init__`` methodand returns the output. Requires jmespath (https://github.com/jmespath/jmespath.py) to run. This processor takes only one input at a time. diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index d253064f2..bf6a02a1d 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -137,7 +137,7 @@ Request objects A string containing the URL of this request. Keep in mind that this attribute contains the escaped URL, so it can differ from the URL passed in - the `__init__` method + the ``__init__`` method This attribute is read-only. To change the URL of a Request use :meth:`replace`. @@ -400,7 +400,7 @@ fields with form data from :class:`Response` objects. .. class:: FormRequest(url, [formdata, ...]) - The :class:`FormRequest` class adds a new keyword parameter to the `__init__` method The + The :class:`FormRequest` class adds a new keyword parameter to the ``__init__`` method The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -473,7 +473,7 @@ fields with form data from :class:`Response` objects. :type dont_click: boolean The other parameters of this class method are passed directly to the - :class:`FormRequest` `__init__` method + :class:`FormRequest` ``__init__`` method .. versionadded:: 0.10.3 The ``formname`` parameter. @@ -547,7 +547,7 @@ dealing with JSON requests. .. class:: JsonRequest(url, [... data, dumps_kwargs]) - The :class:`JsonRequest` class adds two new keyword parameters to the `__init__` method The + The :class:`JsonRequest` class adds two new keyword parameters to the ``__init__`` method The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -721,7 +721,7 @@ TextResponse objects :class:`Response` class, which is meant to be used only for binary data, such as images, sounds or any media file. - :class:`TextResponse` objects support a new `__init__` method argument, in + :class:`TextResponse` objects support a new ``__init__`` method argument, in addition to the base :class:`Response` objects. The remaining functionality is the same as for the :class:`Response` class and is not documented here. @@ -755,7 +755,7 @@ TextResponse objects A string with the encoding of this response. The encoding is resolved by trying the following mechanisms, in order: - 1. the encoding passed in the `__init__` method`` ``encoding`` argument + 1. the encoding passed in the ``__init__`` method`` ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this encoding is not valid (ie. unknown), it is ignored and the next diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 2fdc86b63..8ed8d55f1 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -31,7 +31,7 @@ class BaseItemExporter(object): def _configure(self, options, dont_fail=False): """Configure the exporter by poping options from the ``options`` dict. If dont_fail is set, it won't raise an exception on unexpected options - (useful for using with keyword arguments in subclasses __init__ methods) + (useful for using with keyword arguments in subclasses ``__init__`` methods) """ self.encoding = options.pop('encoding', None) self.fields_to_export = options.pop('fields_to_export', None) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 655c10482..bceb648a0 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -106,7 +106,7 @@ class S3FeedStorage(BlockingFeedStorage): warnings.warn( "Initialising `scrapy.extensions.feedexport.S3FeedStorage` " "without AWS keys is deprecated. Please supply credentials or " - "use the `from_crawler()` __init__ method.", + "use the `from_crawler()` ``__init__`` method.", category=ScrapyDeprecationWarning, stacklevel=2 ) diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index 0dfee1b27..70c2aebc8 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -246,7 +246,7 @@ class CaselessDict(dict): class MergeDict(object): """ A simple class for creating new "virtual" dictionaries that actually look - up values in more than one dictionary, passed in the __init__ method. + up values in more than one dictionary, passed in the ``__init__`` method. If a key appears in more than one of the given dictionaries, only the first occurrence will be used. diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index c892bd438..b3ba2ccec 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -123,14 +123,14 @@ def rel_has_nofollow(rel): def create_instance(objcls, settings, crawler, *args, **kwargs): """Construct a class instance using its ``from_crawler`` or - ``from_settings`` __init__ method, if available. + ``from_settings`` ``__init__`` method, if available. At least one of ``settings`` and ``crawler`` needs to be different from ``None``. If ``settings `` is ``None``, ``crawler.settings`` will be used. - If ``crawler`` is ``None``, only the ``from_settings`` __init__ method will be + If ``crawler`` is ``None``, only the ``from_settings`` ``__init__`` method will be tried. - ``*args`` and ``**kwargs`` are forwarded to the __init__ methods. + ``*args`` and ``**kwargs`` are forwarded to the ``__init__`` methods. Raises ``ValueError`` if both ``settings`` and ``crawler`` are ``None``. """ diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 0d645543a..ea5193f12 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -303,7 +303,7 @@ class WeakKeyCache(object): def stringify_dict(dct_or_tuples, encoding='utf-8', keys_only=True): """Return a (new) dict with unicode keys (and values when "keys_only" is False) of the given dict converted to strings. ``dct_or_tuples`` can be a - dict or a list of tuples, like any dict __init__ method supports. + dict or a list of tuples, like any dict ``__init__`` method supports. """ d = {} for k, v in six.iteritems(dict(dct_or_tuples)): diff --git a/sep/sep-009.rst b/sep/sep-009.rst index 929cedc61..e46479a74 100644 --- a/sep/sep-009.rst +++ b/sep/sep-009.rst @@ -55,7 +55,7 @@ singletons members of that object, as explained below: ``STATS_CLASS`` setting) - **crawler.log**: Logger class with methods replacing the current ``scrapy.log`` functions. Logging would be started (if enabled) on - ``Crawler`` __init__ method, so no log starting functions are required. + ``Crawler`` ``__init__`` method, so no log starting functions are required. - ``crawler.log.msg`` - **crawler.signals**: signal handling diff --git a/tests/test_spider.py b/tests/test_spider.py index 64ff40b61..6f6cdb8ff 100644 --- a/tests/test_spider.py +++ b/tests/test_spider.py @@ -42,12 +42,12 @@ class SpiderTest(unittest.TestCase): self.assertEqual(list(start_requests), []) def test_spider_args(self): - """__init__ method arguments are assigned to spider attributes""" + """``__init__`` method arguments are assigned to spider attributes""" spider = self.spider_class('example.com', foo='bar') self.assertEqual(spider.foo, 'bar') def test_spider_without_name(self): - """__init__ method arguments are assigned to spider attributes""" + """``__init__`` method arguments are assigned to spider attributes""" self.assertRaises(ValueError, self.spider_class) self.assertRaises(ValueError, self.spider_class, somearg='foo') From 85ac5c5c5778a116f762cb6200492aa6a2439e83 Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Mon, 21 Oct 2019 19:06:43 +0100 Subject: [PATCH 024/113] Update .travis.yml Co-Authored-By: Mikhail Korobov --- .travis.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index 2ba504972..044fa9e95 100644 --- a/.travis.yml +++ b/.travis.yml @@ -29,7 +29,7 @@ matrix: python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py37-extra-deps + - env: TOXENV=py38-extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 From 2ee38e8ddbac6841c8d2bb067ddc0ddba813151f Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 14:43:47 +0530 Subject: [PATCH 025/113] Added Pathlib.Path test --- tests/test_feedexport.py | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index f32ac2a4b..842e8d8c0 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -29,6 +29,8 @@ from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get from scrapy.utils.python import to_native_str from scrapy.utils.project import get_project_settings +from Pathlib import Path + class FileFeedStorageTest(unittest.TestCase): @@ -843,3 +845,17 @@ class FeedExportTest(unittest.TestCase): yield self.exported_data({}, settings) self.assertTrue(FromCrawlerCsvItemExporter.init_with_crawler) self.assertTrue(FromCrawlerFileFeedStorage.init_with_crawler) + + @defer.inlineCallbacks + def test_pathlib_uri(self): + tmpdir = tempfile.mkdtemp() + feed_uri = Path(tmpdir) / 'res' + settings = { + 'FEED_FORMAT': 'csv', + 'FEED_STORE_EMPTY': True, + 'FEED_URI': feed_uri, + } + + data = yield self.exported_no_data(settings) + self.assertEqual(data, b'') + shutil.rmtree(tmpdir, ignore_errors=True) \ No newline at end of file From ad96d6ef594c29e8648c3c3b310c42e8fcb210d3 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 14:53:59 +0530 Subject: [PATCH 026/113] Added Pathlib.Path test correctly --- tests/test_feedexport.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 842e8d8c0..664cfd6de 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -856,6 +856,6 @@ class FeedExportTest(unittest.TestCase): 'FEED_URI': feed_uri, } - data = yield self.exported_no_data(settings) - self.assertEqual(data, b'') - shutil.rmtree(tmpdir, ignore_errors=True) \ No newline at end of file + data = yield self.exported_no_data(settings) + self.assertEqual(data, b'') + shutil.rmtree(tmpdir, ignore_errors=True) \ No newline at end of file From 4226791481bb440b9542dda063d1b5ac920a8411 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 15:07:13 +0530 Subject: [PATCH 027/113] Added Pathlib.Path test --- requirements-py2.txt | 1 + requirements-py3.txt | 1 + 2 files changed, 2 insertions(+) diff --git a/requirements-py2.txt b/requirements-py2.txt index dde8d1c9c..42e057417 100644 --- a/requirements-py2.txt +++ b/requirements-py2.txt @@ -16,3 +16,4 @@ service_identity>=16.0.0 six>=1.10.0 Twisted>=16.0.0 zope.interface>=4.1.3 +pathlib2>=2.0 diff --git a/requirements-py3.txt b/requirements-py3.txt index 2c98e6f6d..77296b91b 100644 --- a/requirements-py3.txt +++ b/requirements-py3.txt @@ -16,3 +16,4 @@ lxml>=3.5.0 service_identity>=16.0.0 six>=1.10.0 zope.interface>=4.1.3 +pathlib2>=2.0 From 0b7d8a51b4dabeb4097f5fc2de396904f308c56d Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 15:12:53 +0530 Subject: [PATCH 028/113] Added Pathlib.Path test --- requirements-py2.txt | 2 +- requirements-py3.txt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements-py2.txt b/requirements-py2.txt index 42e057417..2a0bb49d3 100644 --- a/requirements-py2.txt +++ b/requirements-py2.txt @@ -16,4 +16,4 @@ service_identity>=16.0.0 six>=1.10.0 Twisted>=16.0.0 zope.interface>=4.1.3 -pathlib2>=2.0 +pathlib==1.0.1 diff --git a/requirements-py3.txt b/requirements-py3.txt index 77296b91b..c57cff5da 100644 --- a/requirements-py3.txt +++ b/requirements-py3.txt @@ -16,4 +16,4 @@ lxml>=3.5.0 service_identity>=16.0.0 six>=1.10.0 zope.interface>=4.1.3 -pathlib2>=2.0 +pathlib==1.0.1 From a776554282aadc651f9e142aea9bc436f9f587a7 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 15:31:55 +0530 Subject: [PATCH 029/113] Added Pathlib.Path test --- tests/test_feedexport.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 664cfd6de..fcada3a45 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -29,7 +29,7 @@ from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get from scrapy.utils.python import to_native_str from scrapy.utils.project import get_project_settings -from Pathlib import Path +from pathlib import Path class FileFeedStorageTest(unittest.TestCase): From 07822935eca497f3fe31de536afd76a2c31cc5a9 Mon Sep 17 00:00:00 2001 From: illgitthat Date: Tue, 22 Oct 2019 06:05:34 -0400 Subject: [PATCH 030/113] Updating link for miniconda (#4089) --- docs/intro/install.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 2bf98dbdc..51b41b4d7 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -290,5 +290,5 @@ For details, see `Issue #2473 `_. .. _zsh: https://www.zsh.org/ .. _Scrapinghub: https://scrapinghub.com .. _Anaconda: https://docs.anaconda.com/anaconda/ -.. _Miniconda: https://conda.io/docs/user-guide/install/index.html +.. _Miniconda: https://docs.conda.io/projects/conda/en/latest/user-guide/install/index.html .. _conda-forge: https://conda-forge.org/ From cd4c211f4b5c9d5716d0e49472a978c11d7351f8 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 15:38:06 +0530 Subject: [PATCH 031/113] Added Pathlib.Path test --- tests/test_feedexport.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index fcada3a45..f497bb32e 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -29,7 +29,7 @@ from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get from scrapy.utils.python import to_native_str from scrapy.utils.project import get_project_settings -from pathlib import Path +from pathlib2 import Path class FileFeedStorageTest(unittest.TestCase): From 5d75ed4cba3943ac88b4e9cd1d5e87259b85f11f Mon Sep 17 00:00:00 2001 From: WinterComes Date: Tue, 22 Oct 2019 13:19:07 +0300 Subject: [PATCH 032/113] Remove an old note about contracts (#4093) --- docs/topics/contracts.rst | 4 ---- 1 file changed, 4 deletions(-) diff --git a/docs/topics/contracts.rst b/docs/topics/contracts.rst index 62f9a743b..371ae62d5 100644 --- a/docs/topics/contracts.rst +++ b/docs/topics/contracts.rst @@ -6,10 +6,6 @@ Spiders Contracts .. versionadded:: 0.15 -.. note:: This is a new feature (introduced in Scrapy 0.15) and may be subject - to minor functionality/API updates. Check the :ref:`release notes ` to - be notified of updates. - Testing spiders can get particularly annoying and while nothing prevents you from writing unit tests the task gets cumbersome quickly. Scrapy offers an integrated way of testing your spiders by the means of contracts. From cd0964643879b5513807c8955c7e887c77624970 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 16:19:41 +0530 Subject: [PATCH 033/113] Added Pathlib.Path test --- requirements-py2.txt | 2 +- requirements-py3.txt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/requirements-py2.txt b/requirements-py2.txt index 2a0bb49d3..42e057417 100644 --- a/requirements-py2.txt +++ b/requirements-py2.txt @@ -16,4 +16,4 @@ service_identity>=16.0.0 six>=1.10.0 Twisted>=16.0.0 zope.interface>=4.1.3 -pathlib==1.0.1 +pathlib2>=2.0 diff --git a/requirements-py3.txt b/requirements-py3.txt index c57cff5da..77296b91b 100644 --- a/requirements-py3.txt +++ b/requirements-py3.txt @@ -16,4 +16,4 @@ lxml>=3.5.0 service_identity>=16.0.0 six>=1.10.0 zope.interface>=4.1.3 -pathlib==1.0.1 +pathlib2>=2.0 From 7031e3a12422ae977448e7b338457f6c09af9b0e Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 16:31:14 +0530 Subject: [PATCH 034/113] Added Pathlib.Path test --- tests/test_feedexport.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index f497bb32e..c4bfdb4fe 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -850,10 +850,11 @@ class FeedExportTest(unittest.TestCase): def test_pathlib_uri(self): tmpdir = tempfile.mkdtemp() feed_uri = Path(tmpdir) / 'res' + res_uri = urljoin('file:', pathname2url(feed_uri)) settings = { 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, - 'FEED_URI': feed_uri, + 'FEED_URI': res_uri, } data = yield self.exported_no_data(settings) From 85f56a92f0c753dfa55012e647f425a6a3d23076 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 16:43:17 +0530 Subject: [PATCH 035/113] Added Pathlib.Path test --- tests/test_feedexport.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index c4bfdb4fe..17526716b 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -850,6 +850,7 @@ class FeedExportTest(unittest.TestCase): def test_pathlib_uri(self): tmpdir = tempfile.mkdtemp() feed_uri = Path(tmpdir) / 'res' + feed_uri=str(feeduri) res_uri = urljoin('file:', pathname2url(feed_uri)) settings = { 'FEED_FORMAT': 'csv', From 4184bac0687d55a442f599ebcfc913231a7c98e2 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 22 Oct 2019 16:57:14 +0530 Subject: [PATCH 036/113] Added Pathlib.Path test --- tests/test_feedexport.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 17526716b..8c0e5cd3d 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -850,7 +850,7 @@ class FeedExportTest(unittest.TestCase): def test_pathlib_uri(self): tmpdir = tempfile.mkdtemp() feed_uri = Path(tmpdir) / 'res' - feed_uri=str(feeduri) + feed_uri=str(feed_uri) res_uri = urljoin('file:', pathname2url(feed_uri)) settings = { 'FEED_FORMAT': 'csv', From d21e1034f046c7e8d778e698614c47ebd6875c87 Mon Sep 17 00:00:00 2001 From: Ammar Najjar Date: Tue, 22 Oct 2019 13:24:57 +0200 Subject: [PATCH 037/113] docs: correct point,comma and plural replacements Issue #4086 --- docs/topics/exporters.rst | 6 +++--- docs/topics/items.rst | 2 +- docs/topics/loaders.rst | 18 +++++++++--------- docs/topics/request-response.rst | 10 +++++----- scrapy/utils/misc.py | 2 +- 5 files changed, 19 insertions(+), 19 deletions(-) diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index 1b8a69ca3..b8d898022 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -247,7 +247,7 @@ XmlItemExporter :type item_element: str The additional keyword arguments of this ``__init__`` method are passed to the - :class:`BaseItemExporter` ``__init__`` method + :class:`BaseItemExporter` ``__init__`` method. A typical output of this exporter would be:: @@ -352,7 +352,7 @@ PprintItemExporter accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) The additional keyword arguments of this ``__init__`` method are passed to the - :class:`BaseItemExporter` ``__init__`` method + :class:`BaseItemExporter` ``__init__`` method. A typical output of this exporter would be:: @@ -399,7 +399,7 @@ JsonLinesItemExporter Exports Items in JSON format to the specified file-like object, writing one JSON-encoded item per line. The additional ``__init__`` method arguments are passed - to the :class:`BaseItemExporter` ``__init__`` method and the leftover arguments to + to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the `JSONEncoder`_ ``__init__`` method, so you can use any `JSONEncoder`_ ``__init__`` method argument to customize this exporter. diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 370409026..cdf60208e 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -237,7 +237,7 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard `dict API`_, including its ``__init__`` method and + Items replicate the standard `dict API`_, including its ``__init__`` method, and also provide the following additional API members: .. automethod:: copy diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 72610f645..a4465f88d 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -265,7 +265,7 @@ There are several ways to modify Item Loader context values: loader.context['unit'] = 'cm' 2. On Item Loader instantiation (the keyword arguments of Item Loader - ``__init__`` methodare stored in the Item Loader context):: + ``__init__`` method are stored in the Item Loader context):: loader = ItemLoader(product, unit='cm') @@ -494,7 +494,7 @@ ItemLoader objects .. attribute:: default_item_class An Item class (or factory), used to instantiate items when not given in - the ``__init__`` method + the ``__init__`` method. .. attribute:: default_input_processor @@ -509,15 +509,15 @@ ItemLoader objects .. attribute:: default_selector_class The class used to construct the :attr:`selector` of this - :class:`ItemLoader`, if only a response is given in the ``__init__`` method + :class:`ItemLoader`, if only a response is given in the ``__init__`` method. If a selector is given in the ``__init__`` method this attribute is ignored. This attribute is sometimes overridden in subclasses. .. attribute:: selector The :class:`~scrapy.selector.Selector` object to extract data from. - It's either the selector given in the ``__init__`` methodor one created from - the response given in the ``__init__`` methodusing the + It's either the selector given in the ``__init__`` method or one created from + the response given in the ``__init__`` method using the :attr:`default_selector_class`. This attribute is meant to be read-only. @@ -656,7 +656,7 @@ Here is a list of all built-in processors: Returns the first non-null/non-empty value from the values received, so it's typically used as an output processor to single-valued fields. - It doesn't receive any ``__init__`` methodarguments, nor does it accept Loader contexts. + It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts. Example:: @@ -667,7 +667,7 @@ Here is a list of all built-in processors: .. class:: Join(separator=u' ') - Returns the values joined with the separator given in the ``__init__`` method which + Returns the values joined with the separator given in the ``__init__`` method, which defaults to ``u' '``. It doesn't accept Loader contexts. When using the default separator, this processor is equivalent to the @@ -705,7 +705,7 @@ Here is a list of all built-in processors: those which do, this processor will pass the currently active :ref:`Loader context ` through that parameter. - The keyword arguments passed in the ``__init__`` methodare used as the default + The keyword arguments passed in the ``__init__`` method are used as the default Loader context values passed to each function call. However, the final Loader context values passed to functions are overridden with the currently active Loader context accessible through the :meth:`ItemLoader.context` @@ -754,7 +754,7 @@ Here is a list of all built-in processors: .. class:: SelectJmes(json_path) - Queries the value using the json path provided to the ``__init__`` methodand returns the output. + Queries the value using the json path provided to the ``__init__`` method and returns the output. Requires jmespath (https://github.com/jmespath/jmespath.py) to run. This processor takes only one input at a time. diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index bf6a02a1d..123c2dde1 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -137,7 +137,7 @@ Request objects A string containing the URL of this request. Keep in mind that this attribute contains the escaped URL, so it can differ from the URL passed in - the ``__init__`` method + the ``__init__`` method. This attribute is read-only. To change the URL of a Request use :meth:`replace`. @@ -400,7 +400,7 @@ fields with form data from :class:`Response` objects. .. class:: FormRequest(url, [formdata, ...]) - The :class:`FormRequest` class adds a new keyword parameter to the ``__init__`` method The + The :class:`FormRequest` class adds a new keyword parameter to the ``__init__`` method. The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -473,7 +473,7 @@ fields with form data from :class:`Response` objects. :type dont_click: boolean The other parameters of this class method are passed directly to the - :class:`FormRequest` ``__init__`` method + :class:`FormRequest` ``__init__`` method. .. versionadded:: 0.10.3 The ``formname`` parameter. @@ -547,7 +547,7 @@ dealing with JSON requests. .. class:: JsonRequest(url, [... data, dumps_kwargs]) - The :class:`JsonRequest` class adds two new keyword parameters to the ``__init__`` method The + The :class:`JsonRequest` class adds two new keyword parameters to the ``__init__`` method. The remaining arguments are the same as for the :class:`Request` class and are not documented here. @@ -755,7 +755,7 @@ TextResponse objects A string with the encoding of this response. The encoding is resolved by trying the following mechanisms, in order: - 1. the encoding passed in the ``__init__`` method`` ``encoding`` argument + 1. the encoding passed in the ``__init__`` method ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this encoding is not valid (ie. unknown), it is ignored and the next diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index b3ba2ccec..8060553ad 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -123,7 +123,7 @@ def rel_has_nofollow(rel): def create_instance(objcls, settings, crawler, *args, **kwargs): """Construct a class instance using its ``from_crawler`` or - ``from_settings`` ``__init__`` method, if available. + ``from_settings`` ``__init__`` methods, if available. At least one of ``settings`` and ``crawler`` needs to be different from ``None``. If ``settings `` is ``None``, ``crawler.settings`` will be used. From 1d5c270ce8caf954ce83c8db262e2a35707e0c5e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 22 Oct 2019 15:12:52 +0200 Subject: [PATCH 038/113] Fix dangling file descriptor in FeedExporter when FEED_STORE_EMPTY is False (#4023) --- scrapy/extensions/feedexport.py | 4 +++- tests/test_feedexport.py | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index ce2846eba..6fb6397b1 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -242,7 +242,9 @@ class FeedExporter(object): def close_spider(self, spider): slot = self.slot if not slot.itemcount and not self.store_empty: - return + # We need to call slot.storage.store nonetheless to get the file + # properly closed. + return defer.maybeDeferred(slot.storage.store, slot.file) if self._exporting: slot.exporter.finish_exporting() self._exporting = False diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index f32ac2a4b..e1436fbe5 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -417,7 +417,7 @@ class FeedExportTest(unittest.TestCase): content = f.read() finally: - shutil.rmtree(tmpdir, ignore_errors=True) + shutil.rmtree(tmpdir) defer.returnValue(content) From 7a84a4bdba45bab56cee0daed9d9c3a04d8d61c0 Mon Sep 17 00:00:00 2001 From: Ammar Najjar Date: Tue, 22 Oct 2019 15:31:34 +0200 Subject: [PATCH 039/113] docs: use "constructor" for from_settings() & rom_crawler() factory methods Issue #4086 --- scrapy/utils/misc.py | 6 +++--- tests/test_utils_misc/__init__.py | 10 +++++----- 2 files changed, 8 insertions(+), 8 deletions(-) diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index 8060553ad..f638adb25 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -123,14 +123,14 @@ def rel_has_nofollow(rel): def create_instance(objcls, settings, crawler, *args, **kwargs): """Construct a class instance using its ``from_crawler`` or - ``from_settings`` ``__init__`` methods, if available. + ``from_settings`` constructors, if available. At least one of ``settings`` and ``crawler`` needs to be different from ``None``. If ``settings `` is ``None``, ``crawler.settings`` will be used. - If ``crawler`` is ``None``, only the ``from_settings`` ``__init__`` method will be + If ``crawler`` is ``None``, only the ``from_settings`` constructor will be tried. - ``*args`` and ``**kwargs`` are forwarded to the ``__init__`` methods. + ``*args`` and ``**kwargs`` are forwarded to the constructors. Raises ``ValueError`` if both ``settings`` and ``crawler`` are ``None``. """ diff --git a/tests/test_utils_misc/__init__.py b/tests/test_utils_misc/__init__.py index 457a2aa78..e109d5343 100644 --- a/tests/test_utils_misc/__init__.py +++ b/tests/test_utils_misc/__init__.py @@ -109,11 +109,11 @@ class UtilsMiscTestCase(unittest.TestCase): else: mock.assert_called_once_with(*args, **kwargs) - # Check usage of correct __init__ method using four mocks: - # 1. with no alternative __init__ methods - # 2. with from_settings() __init__ method - # 3. with from_crawler() __init__ method - # 4. with from_settings() and from_crawler() __init__ method + # Check usage of correct constructor using four mocks: + # 1. with no alternative constructors + # 2. with from_settings() constructor + # 3. with from_crawler() constructor + # 4. with from_settings() and from_crawler() constructor spec_sets = ([], ['from_settings'], ['from_crawler'], ['from_settings', 'from_crawler']) for specs in spec_sets: From f701f5b0db10faef08e4ed9a21b98fd72f9cfc9a Mon Sep 17 00:00:00 2001 From: Victor Torres Date: Tue, 22 Oct 2019 10:48:02 -0300 Subject: [PATCH 040/113] fix #2552 by improving request schema check on its initialization --- scrapy/http/request/__init__.py | 2 +- tests/test_http_request.py | 2 ++ 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/scrapy/http/request/__init__.py b/scrapy/http/request/__init__.py index d09eaf849..76a428199 100644 --- a/scrapy/http/request/__init__.py +++ b/scrapy/http/request/__init__.py @@ -66,7 +66,7 @@ class Request(object_ref): s = safe_url_string(url, self.encoding) self._url = escape_ajax(s) - if ':' not in self._url: + if ('://' not in self._url) and (not self._url.startswith('data:')): raise ValueError('Missing scheme in request url: %s' % self._url) url = property(_get_url, obsolete_setter(_set_url, 'url')) diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 16d7a1cb8..64f1184c3 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -52,6 +52,8 @@ class RequestTest(unittest.TestCase): def test_url_no_scheme(self): self.assertRaises(ValueError, self.request_class, 'foo') + self.assertRaises(ValueError, self.request_class, '/foo/') + self.assertRaises(ValueError, self.request_class, '/foo:bar') def test_headers(self): # Different ways of setting headers attribute From 7fba8434f3997197f98d2b3e73099ad85429e377 Mon Sep 17 00:00:00 2001 From: Ammar Najjar Date: Tue, 22 Oct 2019 15:55:52 +0200 Subject: [PATCH 041/113] use instantiation for "Crawler" Issue #4086 Co-Authored-By: Mikhail Korobov --- sep/sep-009.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sep/sep-009.rst b/sep/sep-009.rst index e46479a74..da87fa9aa 100644 --- a/sep/sep-009.rst +++ b/sep/sep-009.rst @@ -55,7 +55,7 @@ singletons members of that object, as explained below: ``STATS_CLASS`` setting) - **crawler.log**: Logger class with methods replacing the current ``scrapy.log`` functions. Logging would be started (if enabled) on - ``Crawler`` ``__init__`` method, so no log starting functions are required. + ``Crawler`` instantiation, so no log starting functions are required. - ``crawler.log.msg`` - **crawler.signals**: signal handling From bf5c1a3dec02165b81dffb98fc560c43fb065f38 Mon Sep 17 00:00:00 2001 From: Ammar Najjar Date: Tue, 22 Oct 2019 15:56:46 +0200 Subject: [PATCH 042/113] docs: use "constructor" for "from_crawler" Issue #4086 --- scrapy/extensions/feedexport.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index bceb648a0..ce2846eba 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -106,7 +106,7 @@ class S3FeedStorage(BlockingFeedStorage): warnings.warn( "Initialising `scrapy.extensions.feedexport.S3FeedStorage` " "without AWS keys is deprecated. Please supply credentials or " - "use the `from_crawler()` ``__init__`` method.", + "use the `from_crawler()` constructor.", category=ScrapyDeprecationWarning, stacklevel=2 ) From c623a16a223df31669d9f204b8af7d89062ee56c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 22 Oct 2019 17:52:34 +0200 Subject: [PATCH 043/113] Remove unused method from scrapy.pqueues._SlotPriorityQueues --- scrapy/pqueues.py | 3 --- 1 file changed, 3 deletions(-) diff --git a/scrapy/pqueues.py b/scrapy/pqueues.py index 6ecd1b51a..717ed4d27 100644 --- a/scrapy/pqueues.py +++ b/scrapy/pqueues.py @@ -86,9 +86,6 @@ class _SlotPriorityQueues(object): def __len__(self): return sum(len(x) for x in self.pqueues.values()) if self.pqueues else 0 - def __contains__(self, slot): - return slot in self.pqueues - class ScrapyPriorityQueue(PriorityQueue): """ From 179dc916ff3e3518d8874f3fcd93948e11d5a259 Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Sun, 27 Oct 2019 16:48:20 +0000 Subject: [PATCH 044/113] Update setup.py Updating setup.py to remove python 3.8 support for now --- setup.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/setup.py b/setup.py index 2f5fca4c9..4127d3191 100644 --- a/setup.py +++ b/setup.py @@ -56,7 +56,7 @@ setup( 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', - 'Programming Language :: Python :: 3.8', + # 'Programming Language :: Python :: 3.8', not supported yet 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', From 4068797558f6d8d58f81e12930f487fc0f59f0a5 Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Sun, 27 Oct 2019 17:02:17 +0000 Subject: [PATCH 045/113] Update test_downloadermiddleware_httpcache.py Adding xfail denoting that leveldb is not supported in 3.8 --- tests/test_downloadermiddleware_httpcache.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 22946b98c..34bf5776a 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -154,6 +154,7 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) +@pytest.mark.xfail(reason='leveldb not supported in python 3.8') class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') From deacd34c8d2400b07e911dea9173b840cec7dece Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 17:39:47 +0000 Subject: [PATCH 046/113] [test_downloadermiddleware_httpcache] Attempting to add xfail for leveldb related tests https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 34bf5776a..e350da72c 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -104,6 +104,7 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): + @pytest.mark.xfail(reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -154,7 +155,6 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) -@pytest.mark.xfail(reason='leveldb not supported in python 3.8') class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') From 11942c436c78e998e3d58f0aad7cba490d74cc67 Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:02:13 +0000 Subject: [PATCH 047/113] [test_downloadermiddleware_httpcache] Trying hack to handle systemerror whjen importing leveldb https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index e350da72c..eec0feafc 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,6 +6,7 @@ import unittest import email.utils from contextlib import contextmanager import pytest +import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -157,7 +158,10 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): class LeveldbStorageTest(DefaultStorageTest): - pytest.importorskip('leveldb') + try: + pytest.importorskip('leveldb') + except SystemError: + pass storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From b3df0a84150f15ff00cdfaffc586f91b653baecd Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:28:47 +0000 Subject: [PATCH 048/113] [test_downloadermiddleware_httpcache] Adding xfails to impacted tests following hack fix https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index eec0feafc..605223088 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,7 +6,6 @@ import unittest import email.utils from contextlib import contextmanager import pytest -import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -90,6 +89,7 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) + @pytest.mark.xfail(reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -119,6 +119,7 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None + @pytest.mark.xfail(reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None @@ -161,6 +162,8 @@ class LeveldbStorageTest(DefaultStorageTest): try: pytest.importorskip('leveldb') except SystemError: + # Happens in python 3.8 + # This will cause xfail in DefaultStorageTest to trigger pass storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 70b2854590c1aa10324e6a4b50b6d74a5079c8e9 Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:51:26 +0000 Subject: [PATCH 049/113] [test_downloadermiddleware_httpcache] Making xfails more informative https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 605223088..faee37446 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,6 +6,7 @@ import unittest import email.utils from contextlib import contextmanager import pytest +import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -89,7 +90,7 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) - @pytest.mark.xfail(reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -105,7 +106,7 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): - @pytest.mark.xfail(reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -119,7 +120,7 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None - @pytest.mark.xfail(reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None From 20ea912513053aa2b2122ba5bb0387189bbc5467 Mon Sep 17 00:00:00 2001 From: Roy Date: Sun, 27 Oct 2019 18:52:01 +0000 Subject: [PATCH 050/113] [test_downloadermiddleware_httpcache] Making xfails more informative https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index faee37446..b832fc38c 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -90,7 +90,7 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -106,7 +106,7 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -120,7 +120,7 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=RuntimeError, reason='leveldb not supported in python 3.8') + @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None From bb91f9c78c9c8c892495f6e6252cbeebbe12725b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 21 Oct 2019 12:08:35 +0200 Subject: [PATCH 051/113] Cover Scrapy 1.7.4 in the release notes --- docs/news.rst | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/docs/news.rst b/docs/news.rst index 59317f5eb..8dfe8693c 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -6,6 +6,18 @@ Release notes .. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +Scrapy 1.7.4 (2019-10-21) +------------------------- + +Revert the fix for :issue:`3804` (:issue:`3819`), which has a few undesired +side effects (:issue:`3897`, :issue:`3976`). + +As a result, when an item loader is initialized with an item, +:meth:`ItemLoader.load_item() ` once again +makes later calls to :meth:`ItemLoader.get_output_value() +` or :meth:`ItemLoader.load_item() +` return empty data. + Scrapy 1.7.3 (2019-08-01) ------------------------- From 7731814cc25c57fe31db9ba749450cd5a27eed39 Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 28 Oct 2019 06:53:53 -0300 Subject: [PATCH 052/113] ItemLoader: improve handling of initial item (#4036) --- docs/topics/loaders.rst | 10 +- scrapy/loader/__init__.py | 31 ++-- tests/test_loader.py | 321 +++++++++++++++++++++++++++++--------- 3 files changed, 273 insertions(+), 89 deletions(-) diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 1c2f1da4d..0318e37aa 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -35,6 +35,12 @@ Then, you start collecting values into the Item Loader, typically using the same item field; the Item Loader will know how to "join" those values later using a proper processing function. +.. note:: Collected data is internally stored as lists, + allowing to add several values to the same field. + If an ``item`` argument is passed when creating a loader, + each of the item's values will be stored as-is if it's already + an iterable, or wrapped with a list if it's a single value. + Here is a typical Item Loader usage in a :ref:`Spider `, using the :ref:`Product item ` declared in the :ref:`Items chapter `:: @@ -128,9 +134,9 @@ So what happens is: It's worth noticing that processors are just callable objects, which are called with the data to be parsed, and return a parsed value. So you can use any function as input or output processor. The only requirement is that they must -accept one (and only one) positional argument, which will be an iterator. +accept one (and only one) positional argument, which will be an iterable. -.. note:: Both input and output processors must receive an iterator as their +.. note:: Both input and output processors must receive an iterable as their first argument. The output of those functions can be anything. The result of input processors will be appended to an internal list (in the Loader) containing the collected values (for that field). The result of the output diff --git a/scrapy/loader/__init__.py b/scrapy/loader/__init__.py index 6665eba16..60fd6d222 100644 --- a/scrapy/loader/__init__.py +++ b/scrapy/loader/__init__.py @@ -1,19 +1,19 @@ -"""Item Loader +""" +Item Loader See documentation in docs/topics/loaders.rst - """ from collections import defaultdict + import six from scrapy.item import Item +from scrapy.loader.common import wrap_loader_context +from scrapy.loader.processors import Identity from scrapy.selector import Selector from scrapy.utils.misc import arg_to_iter, extract_regex from scrapy.utils.python import flatten -from .common import wrap_loader_context -from .processors import Identity - class ItemLoader(object): @@ -33,10 +33,9 @@ class ItemLoader(object): self.parent = parent self._local_item = context['item'] = item self._local_values = defaultdict(list) - # Preprocess values if item built from dict - # Values need to be added to item._values if added them from dict (not with add_values) + # values from initial item for field_name, value in item.items(): - self._values[field_name] = self._process_input_value(field_name, value) + self._values[field_name] += arg_to_iter(value) @property def _values(self): @@ -132,8 +131,8 @@ class ItemLoader(object): try: return proc(self._values[field_name]) except Exception as e: - raise ValueError("Error with output processor: field=%r value=%r error='%s: %s'" % \ - (field_name, self._values[field_name], type(e).__name__, str(e))) + raise ValueError("Error with output processor: field=%r value=%r error='%s: %s'" % + (field_name, self._values[field_name], type(e).__name__, str(e))) def get_collected_values(self, field_name): return self._values[field_name] @@ -141,15 +140,15 @@ class ItemLoader(object): def get_input_processor(self, field_name): proc = getattr(self, '%s_in' % field_name, None) if not proc: - proc = self._get_item_field_attr(field_name, 'input_processor', \ - self.default_input_processor) + proc = self._get_item_field_attr(field_name, 'input_processor', + self.default_input_processor) return proc def get_output_processor(self, field_name): proc = getattr(self, '%s_out' % field_name, None) if not proc: - proc = self._get_item_field_attr(field_name, 'output_processor', \ - self.default_output_processor) + proc = self._get_item_field_attr(field_name, 'output_processor', + self.default_output_processor) return proc def _process_input_value(self, field_name, value): @@ -174,8 +173,8 @@ class ItemLoader(object): def _check_selector_method(self): if self.selector is None: raise RuntimeError("To use XPath or CSS selectors, " - "%s must be instantiated with a selector " - "or a response" % self.__class__.__name__) + "%s must be instantiated with a selector " + "or a response" % self.__class__.__name__) def add_xpath(self, field_name, xpath, *processors, **kw): values = self._get_xpathvalues(xpath, **kw) diff --git a/tests/test_loader.py b/tests/test_loader.py index 2725b001a..4a4264a2a 100644 --- a/tests/test_loader.py +++ b/tests/test_loader.py @@ -1,13 +1,15 @@ -import unittest -import six from functools import partial +import unittest + +import six -from scrapy.loader import ItemLoader -from scrapy.loader.processors import Join, Identity, TakeFirst, \ - Compose, MapCompose, SelectJmes -from scrapy.item import Item, Field -from scrapy.selector import Selector from scrapy.http import HtmlResponse +from scrapy.item import Item, Field +from scrapy.loader import ItemLoader +from scrapy.loader.processors import (Compose, Identity, Join, + MapCompose, SelectJmes, TakeFirst) +from scrapy.selector import Selector + # test items class NameItem(Item): @@ -61,7 +63,7 @@ class BasicItemLoaderTest(unittest.TestCase): il.add_value('name', u'marta') item = il.load_item() assert item is i - self.assertEqual(item['summary'], u'lala') + self.assertEqual(item['summary'], [u'lala']) self.assertEqual(item['name'], [u'marta']) def test_load_item_using_custom_loader(self): @@ -419,43 +421,6 @@ class BasicItemLoaderTest(unittest.TestCase): self.assertEqual(item['url'], u'rabbit.hole') self.assertEqual(item['summary'], u'rabbithole') - def test_create_item_from_dict(self): - class TestItem(Item): - title = Field() - - class TestItemLoader(ItemLoader): - default_item_class = TestItem - - input_item = {'title': 'Test item title 1'} - il = TestItemLoader(item=input_item) - # Getting output value mustn't remove value from item - self.assertEqual(il.load_item(), { - 'title': 'Test item title 1', - }) - self.assertEqual(il.get_output_value('title'), 'Test item title 1') - self.assertEqual(il.load_item(), { - 'title': 'Test item title 1', - }) - - input_item = {'title': 'Test item title 2'} - il = TestItemLoader(item=input_item) - # Values from dict must be added to item _values - self.assertEqual(il._values.get('title'), 'Test item title 2') - - input_item = {'title': [u'Test item title 3', u'Test item 4']} - il = TestItemLoader(item=input_item) - # Same rules must work for lists - self.assertEqual(il._values.get('title'), - [u'Test item title 3', u'Test item 4']) - self.assertEqual(il.load_item(), { - 'title': [u'Test item title 3', u'Test item 4'], - }) - self.assertEqual(il.get_output_value('title'), - [u'Test item title 3', u'Test item 4']) - self.assertEqual(il.load_item(), { - 'title': [u'Test item title 3', u'Test item 4'], - }) - def test_error_input_processor(self): class TestItem(Item): name = Field() @@ -493,6 +458,220 @@ class BasicItemLoaderTest(unittest.TestCase): [u'marta', u'other'], Compose(float)) +class InitializationTestMixin(object): + + item_class = None + + def test_keep_single_value(self): + """Loaded item should contain values from the initial item""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo']}) + + def test_keep_list(self): + """Loaded item should contain values from the initial item""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']}) + + def test_add_value_singlevalue_singlevalue(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + il.add_value('name', 'bar') + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar']}) + + def test_add_value_singlevalue_list(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + il.add_value('name', ['item', 'loader']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'item', 'loader']}) + + def test_add_value_list_singlevalue(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + il.add_value('name', 'qwerty') + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'qwerty']}) + + def test_add_value_list_list(self): + """Values added after initialization should be appended""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + il.add_value('name', ['item', 'loader']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(dict(loaded_item), {'name': ['foo', 'bar', 'item', 'loader']}) + + def test_get_output_value_singlevalue(self): + """Getting output value must not remove value from item""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + self.assertEqual(il.get_output_value('name'), ['foo']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(loaded_item, dict({'name': ['foo']})) + + def test_get_output_value_list(self): + """Getting output value must not remove value from item""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + self.assertEqual(il.get_output_value('name'), ['foo', 'bar']) + loaded_item = il.load_item() + self.assertIsInstance(loaded_item, self.item_class) + self.assertEqual(loaded_item, dict({'name': ['foo', 'bar']})) + + def test_values_single(self): + """Values from initial item must be added to loader._values""" + input_item = self.item_class(name='foo') + il = ItemLoader(item=input_item) + self.assertEqual(il._values.get('name'), ['foo']) + + def test_values_list(self): + """Values from initial item must be added to loader._values""" + input_item = self.item_class(name=['foo', 'bar']) + il = ItemLoader(item=input_item) + self.assertEqual(il._values.get('name'), ['foo', 'bar']) + + +class InitializationFromDictTest(InitializationTestMixin, unittest.TestCase): + item_class = dict + + +class InitializationFromItemTest(InitializationTestMixin, unittest.TestCase): + item_class = NameItem + + +class BaseNoInputReprocessingLoader(ItemLoader): + title_in = MapCompose(str.upper) + title_out = TakeFirst() + + +class NoInputReprocessingDictLoader(BaseNoInputReprocessingLoader): + default_item_class = dict + + +class NoInputReprocessingFromDictTest(unittest.TestCase): + """ + Loaders initialized from loaded items must not reprocess fields (dict instances) + """ + def test_avoid_reprocessing_with_initial_values_single(self): + il = NoInputReprocessingDictLoader(item=dict(title='foo')) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='foo')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo')) + + def test_avoid_reprocessing_with_initial_values_list(self): + il = NoInputReprocessingDictLoader(item=dict(title=['foo', 'bar'])) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='foo')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='foo')) + + def test_avoid_reprocessing_without_initial_values_single(self): + il = NoInputReprocessingDictLoader() + il.add_value('title', 'foo') + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='FOO')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO')) + + def test_avoid_reprocessing_without_initial_values_list(self): + il = NoInputReprocessingDictLoader() + il.add_value('title', ['foo', 'bar']) + il_loaded = il.load_item() + self.assertEqual(il_loaded, dict(title='FOO')) + self.assertEqual(NoInputReprocessingDictLoader(item=il_loaded).load_item(), dict(title='FOO')) + + +class NoInputReprocessingItem(Item): + title = Field() + + +class NoInputReprocessingItemLoader(BaseNoInputReprocessingLoader): + default_item_class = NoInputReprocessingItem + + +class NoInputReprocessingFromItemTest(unittest.TestCase): + """ + Loaders initialized from loaded items must not reprocess fields (BaseItem instances) + """ + def test_avoid_reprocessing_with_initial_values_single(self): + il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title='foo')) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'foo'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'}) + + def test_avoid_reprocessing_with_initial_values_list(self): + il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title=['foo', 'bar'])) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'foo'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'foo'}) + + def test_avoid_reprocessing_without_initial_values_single(self): + il = NoInputReprocessingItemLoader() + il.add_value('title', 'FOO') + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'FOO'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'}) + + def test_avoid_reprocessing_without_initial_values_list(self): + il = NoInputReprocessingItemLoader() + il.add_value('title', ['foo', 'bar']) + il_loaded = il.load_item() + self.assertEqual(il_loaded, {'title': 'FOO'}) + self.assertEqual(NoInputReprocessingItemLoader(item=il_loaded).load_item(), {'title': 'FOO'}) + + +class TestOutputProcessorDict(unittest.TestCase): + def test_output_processor(self): + + class TempDict(dict): + def __init__(self, *args, **kwargs): + super(TempDict, self).__init__(self, *args, **kwargs) + self.setdefault('temp', 0.3) + + class TempLoader(ItemLoader): + default_item_class = TempDict + default_input_processor = Identity() + default_output_processor = Compose(TakeFirst()) + + loader = TempLoader() + item = loader.load_item() + self.assertIsInstance(item, TempDict) + self.assertEqual(dict(item), {'temp': 0.3}) + + +class TestOutputProcessorItem(unittest.TestCase): + def test_output_processor(self): + + class TempItem(Item): + temp = Field() + + def __init__(self, *args, **kwargs): + super(TempItem, self).__init__(self, *args, **kwargs) + self.setdefault('temp', 0.3) + + class TempLoader(ItemLoader): + default_item_class = TempItem + default_input_processor = Identity() + default_output_processor = Compose(TakeFirst()) + + loader = TempLoader() + item = loader.load_item() + self.assertIsInstance(item, TempItem) + self.assertEqual(dict(item), {'temp': 0.3}) + + class ProcessorsTest(unittest.TestCase): def test_take_first(self): @@ -523,7 +702,8 @@ class ProcessorsTest(unittest.TestCase): self.assertRaises(ValueError, proc, 'hello') def test_mapcompose(self): - filter_world = lambda x: None if x == 'world' else x + def filter_world(x): + return None if x == 'world' else x proc = MapCompose(filter_world, six.text_type.upper) self.assertEqual(proc([u'hello', u'world', u'this', u'is', u'scrapy']), [u'HELLO', u'THIS', u'IS', u'SCRAPY']) @@ -535,7 +715,6 @@ class ProcessorsTest(unittest.TestCase): self.assertRaises(ValueError, proc, 'hello') - class SelectortemLoaderTest(unittest.TestCase): response = HtmlResponse(url="", encoding='utf-8', body=b""" @@ -672,7 +851,7 @@ class SelectortemLoaderTest(unittest.TestCase): self.assertEqual(l.get_css(['p::text', 'div::text']), [u'paragraph', 'marta']) self.assertEqual(l.get_css(['a::attr(href)', 'img::attr(src)']), - [u'http://www.scrapy.org', u'/images/logo.png']) + [u'http://www.scrapy.org', u'/images/logo.png']) def test_replace_css_multi_fields(self): l = TestItemLoader(response=self.response) @@ -720,7 +899,7 @@ class SubselectorLoaderTest(unittest.TestCase): self.assertEqual(l.get_output_value('name'), [u'marta']) self.assertEqual(l.get_output_value('name_div'), [u'
marta
']) - self.assertEqual(l.get_output_value('name_value'), [u'marta']) + self.assertEqual(l.get_output_value('name_value'), [u'marta']) self.assertEqual(l.get_output_value('name'), nl.get_output_value('name')) self.assertEqual(l.get_output_value('name_div'), nl.get_output_value('name_div')) @@ -735,7 +914,7 @@ class SubselectorLoaderTest(unittest.TestCase): self.assertEqual(l.get_output_value('name'), [u'marta']) self.assertEqual(l.get_output_value('name_div'), [u'
marta
']) - self.assertEqual(l.get_output_value('name_value'), [u'marta']) + self.assertEqual(l.get_output_value('name_value'), [u'marta']) self.assertEqual(l.get_output_value('name'), nl.get_output_value('name')) self.assertEqual(l.get_output_value('name_div'), nl.get_output_value('name_div')) @@ -791,28 +970,28 @@ class SubselectorLoaderTest(unittest.TestCase): class SelectJmesTestCase(unittest.TestCase): - test_list_equals = { - 'simple': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), - 'invalid': ('foo.bar.baz', {"foo": {"bar": "baz"}}, None), - 'top_level': ('foo', {"foo": {"bar": "baz"}}, {"bar": "baz"}), - 'double_vs_single_quote_string': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), - 'dict': ( - 'foo.bar[*].name', - {"foo": {"bar": [{"name": "one"}, {"name": "two"}]}}, - ['one', 'two'] - ), - 'list': ('[1]', [1, 2], 2) - } + test_list_equals = { + 'simple': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), + 'invalid': ('foo.bar.baz', {"foo": {"bar": "baz"}}, None), + 'top_level': ('foo', {"foo": {"bar": "baz"}}, {"bar": "baz"}), + 'double_vs_single_quote_string': ('foo.bar', {"foo": {"bar": "baz"}}, "baz"), + 'dict': ( + 'foo.bar[*].name', + {"foo": {"bar": [{"name": "one"}, {"name": "two"}]}}, + ['one', 'two'] + ), + 'list': ('[1]', [1, 2], 2) + } - def test_output(self): - for l in self.test_list_equals: - expr, test_list, expected = self.test_list_equals[l] - test = SelectJmes(expr)(test_list) - self.assertEqual( - test, - expected, - msg='test "{}" got {} expected {}'.format(l, test, expected) - ) + def test_output(self): + for l in self.test_list_equals: + expr, test_list, expected = self.test_list_equals[l] + test = SelectJmes(expr)(test_list) + self.assertEqual( + test, + expected, + msg='test "{}" got {} expected {}'.format(l, test, expected) + ) if __name__ == "__main__": From b5a00262ec48534b89750037060318326b4e349c Mon Sep 17 00:00:00 2001 From: Roy Healy Date: Mon, 28 Oct 2019 09:59:33 +0000 Subject: [PATCH 053/113] Update .travis.yml Reverting change to 3.8 extra dependency environment. --- .travis.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index 044fa9e95..2ba504972 100644 --- a/.travis.yml +++ b/.travis.yml @@ -29,7 +29,7 @@ matrix: python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py38-extra-deps + - env: TOXENV=py37-extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 From 3d0df419c4eeb0ccf7934c8f06a38707eae8d722 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 28 Oct 2019 11:24:47 +0100 Subject: [PATCH 054/113] Mark the LevelDB storage backend as deprecated --- docs/topics/downloader-middleware.rst | 22 ---------------------- scrapy/extensions/httpcache.py | 17 ++++++++++++----- 2 files changed, 12 insertions(+), 27 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 2892b9b79..8048e1c86 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -348,7 +348,6 @@ HttpCacheMiddleware * :ref:`httpcache-storage-fs` * :ref:`httpcache-storage-dbm` - * :ref:`httpcache-storage-leveldb` You can change the HTTP cache storage backend with the :setting:`HTTPCACHE_STORAGE` setting. Or you can also :ref:`implement your own storage backend. ` @@ -478,27 +477,6 @@ DBM storage backend By default, it uses the anydbm_ module, but you can change it with the :setting:`HTTPCACHE_DBM_MODULE` setting. -.. _httpcache-storage-leveldb: - -LevelDB storage backend -~~~~~~~~~~~~~~~~~~~~~~~ - -.. class:: LeveldbCacheStorage - - .. versionadded:: 0.23 - - A LevelDB_ storage backend is also available for the HTTP cache middleware. - - This backend is not recommended for development because only one process - can access LevelDB databases at the same time, so you can't run a crawl and - open the scrapy shell in parallel for the same spider. - - In order to use this storage backend, install the `LevelDB python - bindings`_ (e.g. ``pip install leveldb``). - - .. _LevelDB: https://github.com/google/leveldb - .. _leveldb python bindings: https://pypi.python.org/pypi/leveldb - .. _httpcache-storage-custom: Writing your own storage backend diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index c6094643d..7c650a91e 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -1,19 +1,24 @@ from __future__ import print_function -import os + import gzip import logging -from six.moves import cPickle as pickle +import os +from email.utils import mktime_tz, parsedate_tz from importlib import import_module from time import time +from warnings import warn from weakref import WeakKeyDictionary -from email.utils import mktime_tz, parsedate_tz + +from six.moves import cPickle as pickle from w3lib.http import headers_raw_to_dict, headers_dict_to_raw + +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Headers, Response from scrapy.responsetypes import responsetypes -from scrapy.utils.request import request_fingerprint -from scrapy.utils.project import data_path from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.project import data_path from scrapy.utils.python import to_bytes, to_unicode, garbage_collect +from scrapy.utils.request import request_fingerprint logger = logging.getLogger(__name__) @@ -345,6 +350,8 @@ class FilesystemCacheStorage(object): class LeveldbCacheStorage(object): def __init__(self, settings): + warn("The LevelDB storage backend is deprecated.", + ScrapyDeprecationWarning, stacklevel=2) import leveldb self._leveldb = leveldb self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True) From 16bb3ac20dae8b7c5fbccf4ab85b3a0393e7c55d Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 11:24:09 +0000 Subject: [PATCH 055/113] [test_downloadermiddleware_httpcache] Using skipif approach https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index b832fc38c..ba5027307 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -90,7 +90,6 @@ class _BaseTest(unittest.TestCase): assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_dont_cache(self): with self._middleware() as mw: self.request.meta['dont_cache'] = True @@ -106,7 +105,6 @@ class _BaseTest(unittest.TestCase): class DefaultStorageTest(_BaseTest): - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage(self): with self._storage() as storage: request2 = self.request.copy() @@ -120,7 +118,6 @@ class DefaultStorageTest(_BaseTest): time.sleep(2) # wait for cache to expire assert storage.retrieve_response(self.spider, request2) is None - @pytest.mark.xfail(sys.version_info >= (3, 8), raises=SystemError, reason='leveldb not supported in python 3.8') def test_storage_never_expire(self): with self._storage(HTTPCACHE_EXPIRATION_SECS=0) as storage: assert storage.retrieve_response(self.spider, self.request) is None @@ -164,8 +161,10 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - # This will cause xfail in DefaultStorageTest to trigger - pass + pytest.mark.skipif( + sys.version_info >= (3, 8), + reason='leveldb not supported in python 3.8', + ) storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 9b47dc6a703310d13c9470e50d4b14f81ee893c6 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 11:24:52 +0000 Subject: [PATCH 056/113] [travis, setup] Adding official python 3.8 support https://github.com/scrapy/scrapy/issues/4085 --- .travis.yml | 4 +--- setup.py | 2 +- 2 files changed, 2 insertions(+), 4 deletions(-) diff --git a/.travis.yml b/.travis.yml index 2ba504972..4c2498053 100644 --- a/.travis.yml +++ b/.travis.yml @@ -25,11 +25,9 @@ matrix: python: 3.6 - env: TOXENV=py37 python: 3.7 - - env: TOXENV=py37-extra-deps - python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py37-extra-deps + - env: TOXENV=py38-extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 diff --git a/setup.py b/setup.py index 4127d3191..2f5fca4c9 100644 --- a/setup.py +++ b/setup.py @@ -56,7 +56,7 @@ setup( 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', 'Programming Language :: Python :: 3.7', - # 'Programming Language :: Python :: 3.8', not supported yet + 'Programming Language :: Python :: 3.8', 'Programming Language :: Python :: Implementation :: CPython', 'Programming Language :: Python :: Implementation :: PyPy', 'Topic :: Internet :: WWW/HTTP', From 4432136ffff4d8af42f7a485c17ab7fbbb228078 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:22:21 +0000 Subject: [PATCH 057/113] [test_downloadermiddleware_httpcache] Fixing pytest skip behaviour https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index ba5027307..32085b095 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,7 +6,6 @@ import unittest import email.utils from contextlib import contextmanager import pytest -import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -161,10 +160,7 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - pytest.mark.skipif( - sys.version_info >= (3, 8), - reason='leveldb not supported in python 3.8', - ) + pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From c51fb959e2985faf6f21fe7f03d2fb8160de064f Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:36:54 +0000 Subject: [PATCH 058/113] [test_downloadermiddleware_httpcache] Fixing pytest skip behaviour https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 32085b095..047526569 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -6,6 +6,7 @@ import unittest import email.utils from contextlib import contextmanager import pytest +import sys from scrapy.http import Response, HtmlResponse, Request from scrapy.spiders import Spider @@ -160,7 +161,7 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") + pytestmark = pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 74909030a55b59e3b858fc736b5b1f685d9596a6 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:52:36 +0000 Subject: [PATCH 059/113] [tox.ini] Removing obsolete py37 extra deps enviornment https://github.com/scrapy/scrapy/issues/4085 --- tox.ini | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tox.ini b/tox.ini index 14afec23f..fe925951b 100644 --- a/tox.ini +++ b/tox.ini @@ -125,13 +125,6 @@ deps = {[docs]deps} commands = sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck -[testenv:py37-extra-deps] -basepython = python3.7 -deps = - {[testenv:py35]deps} - reppy - robotexclusionrulesparser - [testenv:py38-extra-deps] basepython = python3.8 deps = From b73d217de5647a68c7b8dfda747cd3d0685c226d Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 12:55:54 +0000 Subject: [PATCH 060/113] [test_downloadermiddleware_httpcache.py] Fixing pytest mark behaviour https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 047526569..f5917d0f0 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -161,7 +161,7 @@ class LeveldbStorageTest(DefaultStorageTest): pytest.importorskip('leveldb') except SystemError: # Happens in python 3.8 - pytestmark = pytest.skip("'SystemError: bad call flags' occurs on Python 3.8") + pytestmark = pytest.mark.skip("'SystemError: bad call flags' occurs on Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 93e3dc1b826e44d1a5a24fbb39c090ce426aa862 Mon Sep 17 00:00:00 2001 From: Roy Date: Mon, 28 Oct 2019 16:12:03 +0000 Subject: [PATCH 061/113] [test_downloadermiddleware_httpcache.py] Cleaning text https://github.com/scrapy/scrapy/issues/4085 --- tests/test_downloadermiddleware_httpcache.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index f5917d0f0..972d400a4 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -155,13 +155,13 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): new_settings.setdefault('HTTPCACHE_GZIP', True) return super(FilesystemStorageTest, self)._get_settings(**new_settings) + class LeveldbStorageTest(DefaultStorageTest): try: pytest.importorskip('leveldb') except SystemError: - # Happens in python 3.8 - pytestmark = pytest.mark.skip("'SystemError: bad call flags' occurs on Python 3.8") + pytestmark = pytest.mark.skip("Test module skipped - 'SystemError: bad call flags' occurs when >= Python 3.8") storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' From 94f060fcc84853f28f3f91b6dde1d61c8e19251e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 29 Oct 2019 12:53:46 +0100 Subject: [PATCH 062/113] Cover Scrapy 1.8.0 in the release notes (#3952) --- docs/news.rst | 226 +++++++++++++++++++++++++++++++++++++++- docs/topics/logging.rst | 5 +- scrapy/logformatter.py | 2 +- 3 files changed, 229 insertions(+), 4 deletions(-) diff --git a/docs/news.rst b/docs/news.rst index 8dfe8693c..669844045 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -6,6 +6,209 @@ Release notes .. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +.. _release-1.8.0: + +Scrapy 1.8.0 (2019-10-28) +------------------------- + +Highlights: + +* Dropped Python 3.4 support and updated minimum requirements; made Python 3.8 + support official +* New :meth:`Request.from_curl ` class method +* New :setting:`ROBOTSTXT_PARSER` and :setting:`ROBOTSTXT_USER_AGENT` settings +* New :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` and + :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` settings + +Backward-incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Python 3.4 is no longer supported, and some of the minimum requirements of + Scrapy have also changed: + + * cssselect_ 0.9.1 + * cryptography_ 2.0 + * lxml_ 3.5.0 + * pyOpenSSL_ 16.2.0 + * queuelib_ 1.4.2 + * service_identity_ 16.0.0 + * six_ 1.10.0 + * Twisted_ 17.9.0 (16.0.0 with Python 2) + * zope.interface_ 4.1.3 + + (:issue:`3892`) + +* ``JSONRequest`` is now called :class:`~scrapy.http.JsonRequest` for + consistency with similar classes (:issue:`3929`, :issue:`3982`) + +* If you are using a custom context factory + (:setting:`DOWNLOADER_CLIENTCONTEXTFACTORY`), its ``__init__`` method must + accept two new parameters: ``tls_verbose_logging`` and ``tls_ciphers`` + (:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`) + +* :class:`~scrapy.loader.ItemLoader` now turns the values of its input item + into lists:: + + >>> item = MyItem() + >>> item['field'] = 'value1' + >>> loader = ItemLoader(item=item) + >>> item['field'] + ['value1'] + + This is needed to allow adding values to existing fields + (``loader.add_value('field', 'value2')``). + + (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, :issue:`3998`, + :issue:`4036`) + +See also :ref:`1.8-deprecation-removals` below. + + +New features +~~~~~~~~~~~~ + +* A new :meth:`Request.from_curl ` class + method allows :ref:`creating a request from a cURL command + ` (:issue:`2985`, :issue:`3862`) + +* A new :setting:`ROBOTSTXT_PARSER` setting allows choosing which robots.txt_ + parser to use. It includes built-in support for + :ref:`RobotFileParser `, + :ref:`Protego ` (default), :ref:`Reppy `, and + :ref:`Robotexclusionrulesparser `, and allows you to + :ref:`implement support for additional parsers + ` (:issue:`754`, :issue:`2669`, + :issue:`3796`, :issue:`3935`, :issue:`3969`, :issue:`4006`) + +* A new :setting:`ROBOTSTXT_USER_AGENT` setting allows defining a separate + user agent string to use for robots.txt_ parsing (:issue:`3931`, + :issue:`3966`) + +* :class:`~scrapy.spiders.Rule` no longer requires a :class:`LinkExtractor + ` parameter + (:issue:`781`, :issue:`4016`) + +* Use the new :setting:`DOWNLOADER_CLIENT_TLS_CIPHERS` setting to customize + the TLS/SSL ciphers used by the default HTTP/1.1 downloader (:issue:`3392`, + :issue:`3442`) + +* Set the new :setting:`DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING` setting to + ``True`` to enable debug-level messages about TLS connection parameters + after establishing HTTPS connections (:issue:`2111`, :issue:`3450`) + +* Callbacks that receive keyword arguments + (see :attr:`Request.cb_kwargs `) can now be + tested using the new :class:`@cb_kwargs + ` + :ref:`spider contract ` (:issue:`3985`, :issue:`3988`) + +* When a :class:`@scrapes ` spider + contract fails, all missing fields are now reported (:issue:`766`, + :issue:`3939`) + +* :ref:`Custom log formats ` can now drop messages by + having the corresponding methods of the configured :setting:`LOG_FORMATTER` + return ``None`` (:issue:`3984`, :issue:`3987`) + +* A much improved completion definition is now available for Zsh_ + (:issue:`4069`) + + +Bug fixes +~~~~~~~~~ + +* :meth:`ItemLoader.load_item() ` no + longer makes later calls to :meth:`ItemLoader.get_output_value() + ` or + :meth:`ItemLoader.load_item() ` return + empty data (:issue:`3804`, :issue:`3819`, :issue:`3897`, :issue:`3976`, + :issue:`3998`, :issue:`4036`) + +* Fixed :class:`~scrapy.statscollectors.DummyStatsCollector` raising a + :exc:`TypeError` exception (:issue:`4007`, :issue:`4052`) + +* :meth:`FilesPipeline.file_path + ` and + :meth:`ImagesPipeline.file_path + ` no longer choose + file extensions that are not `registered with IANA`_ (:issue:`1287`, + :issue:`3953`, :issue:`3954`) + +* When using botocore_ to persist files in S3, all botocore-supported headers + are properly mapped now (:issue:`3904`, :issue:`3905`) + +* FTP passwords in :setting:`FEED_URI` containing percent-escaped characters + are now properly decoded (:issue:`3941`) + +* A memory-handling and error-handling issue in + :func:`scrapy.utils.ssl.get_temp_key_info` has been fixed (:issue:`3920`) + + +Documentation +~~~~~~~~~~~~~ + +* The documentation now covers how to define and configure a :ref:`custom log + format ` (:issue:`3616`, :issue:`3660`) + +* API documentation added for :class:`~scrapy.exporters.MarshalItemExporter` + and :class:`~scrapy.exporters.PythonItemExporter` (:issue:`3973`) + +* API documentation added for :class:`~scrapy.item.BaseItem` and + :class:`~scrapy.item.ItemMeta` (:issue:`3999`) + +* Minor documentation fixes (:issue:`2998`, :issue:`3398`, :issue:`3597`, + :issue:`3894`, :issue:`3934`, :issue:`3978`, :issue:`3993`, :issue:`4022`, + :issue:`4028`, :issue:`4033`, :issue:`4046`, :issue:`4050`, :issue:`4055`, + :issue:`4056`, :issue:`4061`, :issue:`4072`, :issue:`4071`, :issue:`4079`, + :issue:`4081`, :issue:`4089`, :issue:`4093`) + + +.. _1.8-deprecation-removals: + +Deprecation removals +~~~~~~~~~~~~~~~~~~~~ + +* ``scrapy.xlib`` has been removed (:issue:`4015`) + + +Deprecations +~~~~~~~~~~~~ + +* The LevelDB_ storage backend + (``scrapy.extensions.httpcache.LeveldbCacheStorage``) of + :class:`~scrapy.downloadermiddlewares.httpcache.HttpCacheMiddleware` is + deprecated (:issue:`4085`, :issue:`4092`) + +* Use of the undocumented ``SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE`` environment + variable is deprecated (:issue:`3910`) + +* ``scrapy.item.DictItem`` is deprecated, use :class:`~scrapy.item.Item` + instead (:issue:`3999`) + + +Other changes +~~~~~~~~~~~~~ + +* Minimum versions of optional Scrapy requirements that are covered by + continuous integration tests have been updated: + + * botocore_ 1.3.23 + * Pillow_ 3.4.2 + + Lower versions of these optional requirements may work, but it is not + guaranteed (:issue:`3892`) + +* GitHub templates for bug reports and feature requests (:issue:`3126`, + :issue:`3471`, :issue:`3749`, :issue:`3754`) + +* Continuous integration fixes (:issue:`3923`) + +* Code cleanup (:issue:`3391`, :issue:`3907`, :issue:`3946`, :issue:`3950`, + :issue:`4023`, :issue:`4031`) + + +.. _release-1.7.4: + Scrapy 1.7.4 (2019-10-21) ------------------------- @@ -18,22 +221,31 @@ makes later calls to :meth:`ItemLoader.get_output_value() ` or :meth:`ItemLoader.load_item() ` return empty data. + +.. _release-1.7.3: + Scrapy 1.7.3 (2019-08-01) ------------------------- Enforce lxml 4.3.5 or lower for Python 3.4 (:issue:`3912`, :issue:`3918`). + +.. _release-1.7.2: + Scrapy 1.7.2 (2019-07-23) ------------------------- Fix Python 2 support (:issue:`3889`, :issue:`3893`, :issue:`3896`). +.. _release-1.7.1: + Scrapy 1.7.1 (2019-07-18) ------------------------- Re-packaging of Scrapy 1.7.0, which was missing some changes in PyPI. + .. _release-1.7.0: Scrapy 1.7.0 (2019-07-18) @@ -568,7 +780,7 @@ Scrapy 1.5.2 (2019-01-22) See :ref:`telnet console ` documentation for more info -* Backport CI build failure under GCE environemnt due to boto import error. +* Backport CI build failure under GCE environment due to boto import error. .. _release-1.5.1: @@ -2830,23 +3042,35 @@ First release of Scrapy. .. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1 +.. _botocore: https://github.com/boto/botocore .. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding .. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/ .. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request +.. _cryptography: https://cryptography.io/en/latest/ .. _cssselect: https://github.com/scrapy/cssselect/ .. _docstrings: https://docs.python.org/glossary.html#term-docstring .. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt +.. _LevelDB: https://github.com/google/leveldb .. _lxml: http://lxml.de/ .. _marshal: https://docs.python.org/2/library/marshal.html .. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator .. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator .. _parsel.csstranslator.XPathExpr: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.XPathExpr .. _PEP 257: https://www.python.org/dev/peps/pep-0257/ +.. _Pillow: https://python-pillow.org/ +.. _pyOpenSSL: https://www.pyopenssl.org/en/stable/ .. _queuelib: https://github.com/scrapy/queuelib +.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml .. _resource: https://docs.python.org/2/library/resource.html +.. _robots.txt: http://www.robotstxt.org/ .. _scrapely: https://github.com/scrapy/scrapely +.. _service_identity: https://service-identity.readthedocs.io/en/stable/ +.. _six: https://six.readthedocs.io/ .. _tox: https://pypi.python.org/pypi/tox +.. _Twisted: https://twistedmatrix.com/trac/ .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _w3lib: https://github.com/scrapy/w3lib .. _w3lib.encoding: https://github.com/scrapy/w3lib/blob/master/w3lib/encoding.py .. _What is cacheable: https://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9.1 +.. _zope.interface: https://zopeinterface.readthedocs.io/en/latest/ +.. _Zsh: https://www.zsh.org/ diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index 87ea43c7d..2db0ffddd 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -198,8 +198,9 @@ to override some of the Scrapy settings regarding logging. Custom Log Formats ------------------ -A custom log format can be set for different actions by extending :class:`~scrapy.logformatter.LogFormatter` class -and making :setting:`LOG_FORMATTER` point to your new class. +A custom log format can be set for different actions by extending +:class:`~scrapy.logformatter.LogFormatter` class and making +:setting:`LOG_FORMATTER` point to your new class. .. autoclass:: scrapy.logformatter.LogFormatter :members: diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index f15940ed1..3c61ed7e0 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -29,7 +29,7 @@ class LogFormatter(object): * ``args`` should be a tuple or dict with the formatting placeholders for ``msg``. The final log message is computed as ``msg % args``. - Users can define their own ``LogFormatter`` class if they want to customise how + Users can define their own ``LogFormatter`` class if they want to customize how each action is logged or if they want to omit it entirely. In order to omit logging an action the method must return ``None``. From be2e910dd06ba4904e7b10eb5a7e3251e8dab099 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 29 Oct 2019 12:57:02 +0100 Subject: [PATCH 063/113] =?UTF-8?q?Bump=20version:=201.7.0=20=E2=86=92=201?= =?UTF-8?q?.8.0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.cfg | 2 +- scrapy/VERSION | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/.bumpversion.cfg b/.bumpversion.cfg index 70affe63f..c9f1abea5 100644 --- a/.bumpversion.cfg +++ b/.bumpversion.cfg @@ -1,5 +1,5 @@ [bumpversion] -current_version = 1.7.0 +current_version = 1.8.0 commit = True tag = True tag_name = {new_version} diff --git a/scrapy/VERSION b/scrapy/VERSION index bd8bf882d..27f9cd322 100644 --- a/scrapy/VERSION +++ b/scrapy/VERSION @@ -1 +1 @@ -1.7.0 +1.8.0 From 66cbceeb0a9104fc0fa238898e38d0d9ce9cbcf6 Mon Sep 17 00:00:00 2001 From: Amardeep Bhowmick Date: Wed, 30 Oct 2019 13:39:12 +0530 Subject: [PATCH 064/113] Fix redirection error when the Location header value starts with 3 slashes (#4042) --- scrapy/downloadermiddlewares/redirect.py | 7 +++++-- tests/test_downloadermiddleware_redirect.py | 16 ++++++++++++++++ 2 files changed, 21 insertions(+), 2 deletions(-) diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 49468a2e4..b73f864dd 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -1,5 +1,5 @@ import logging -from six.moves.urllib.parse import urljoin +from six.moves.urllib.parse import urljoin, urlparse from w3lib.url import safe_url_string @@ -70,7 +70,10 @@ class RedirectMiddleware(BaseRedirectMiddleware): if 'Location' not in response.headers or response.status not in allowed_status: return response - location = safe_url_string(response.headers['location']) + location = safe_url_string(response.headers['Location']) + if response.headers['Location'].startswith(b'//'): + request_scheme = urlparse(request.url).scheme + location = request_scheme + '://' + location.lstrip('/') redirected_url = urljoin(request.url, location) diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index 0e841489d..e7faf14a7 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -106,6 +106,22 @@ class RedirectMiddlewareTest(unittest.TestCase): del rsp.headers['Location'] assert self.mw.process_response(req, rsp, self.spider) is rsp + def test_redirect_302_relative(self): + url = 'http://www.example.com/302' + url2 = '///i8n.example2.com/302' + url3 = 'http://i8n.example2.com/302' + req = Request(url, method='HEAD') + rsp = Response(url, headers={'Location': url2}, status=302) + + req2 = self.mw.process_response(req, rsp, self.spider) + assert isinstance(req2, Request) + self.assertEqual(req2.url, url3) + self.assertEqual(req2.method, 'HEAD') + + # response without Location header but with status code is 3XX should be ignored + del rsp.headers['Location'] + assert self.mw.process_response(req, rsp, self.spider) is rsp + def test_max_redirect_times(self): self.mw.max_redirect_times = 1 From 6d6da78eda3cc0bba1bfdf70194fdf655fac8aeb Mon Sep 17 00:00:00 2001 From: Benjamin Ooghe-Tabanou Date: Wed, 30 Oct 2019 09:13:36 +0100 Subject: [PATCH 065/113] Add a keep_fragments parameter to the request_fingerprint function (#4104) --- scrapy/utils/request.py | 16 +++++++++++----- tests/test_utils_request.py | 9 ++++++++- 2 files changed, 19 insertions(+), 6 deletions(-) diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index 9c143b83a..fb5af66a2 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -16,7 +16,7 @@ from scrapy.utils.httpobj import urlparse_cached _fingerprint_cache = weakref.WeakKeyDictionary() -def request_fingerprint(request, include_headers=None): +def request_fingerprint(request, include_headers=None, keep_fragments=False): """ Return the request fingerprint. @@ -42,15 +42,21 @@ def request_fingerprint(request, include_headers=None): the fingeprint. If you want to include specific headers use the include_headers argument, which is a list of Request headers to include. + Also, servers usually ignore fragments in urls when handling requests, + so they are also ignored by default when calculating the fingerprint. + If you want to include them, set the keep_fragments argument to True + (for instance when handling requests with a headless browser). + """ if include_headers: include_headers = tuple(to_bytes(h.lower()) for h in sorted(include_headers)) cache = _fingerprint_cache.setdefault(request, {}) - if include_headers not in cache: + cache_key = (include_headers, keep_fragments) + if cache_key not in cache: fp = hashlib.sha1() fp.update(to_bytes(request.method)) - fp.update(to_bytes(canonicalize_url(request.url))) + fp.update(to_bytes(canonicalize_url(request.url, keep_fragments=keep_fragments))) fp.update(request.body or b'') if include_headers: for hdr in include_headers: @@ -58,8 +64,8 @@ def request_fingerprint(request, include_headers=None): fp.update(hdr) for v in request.headers.getlist(hdr): fp.update(v) - cache[include_headers] = fp.hexdigest() - return cache[include_headers] + cache[cache_key] = fp.hexdigest() + return cache[cache_key] def request_authenticate(request, username, password): diff --git a/tests/test_utils_request.py b/tests/test_utils_request.py index e8a4eb3ea..625a32048 100644 --- a/tests/test_utils_request.py +++ b/tests/test_utils_request.py @@ -17,7 +17,7 @@ class UtilsRequestTest(unittest.TestCase): self.assertNotEqual(request_fingerprint(r1), request_fingerprint(r2)) # make sure caching is working - self.assertEqual(request_fingerprint(r1), _fingerprint_cache[r1][None]) + self.assertEqual(request_fingerprint(r1), _fingerprint_cache[r1][(None, False)]) r1 = Request("http://www.example.com/members/offers.html") r2 = Request("http://www.example.com/members/offers.html") @@ -42,6 +42,13 @@ class UtilsRequestTest(unittest.TestCase): self.assertEqual(request_fingerprint(r3, include_headers=['accept-language', 'sessionid']), request_fingerprint(r3, include_headers=['SESSIONID', 'Accept-Language'])) + r1 = Request("http://www.example.com/test.html") + r2 = Request("http://www.example.com/test.html#fragment") + self.assertEqual(request_fingerprint(r1), request_fingerprint(r2)) + self.assertEqual(request_fingerprint(r1), request_fingerprint(r1, keep_fragments=True)) + self.assertNotEqual(request_fingerprint(r2), request_fingerprint(r2, keep_fragments=True)) + self.assertNotEqual(request_fingerprint(r1), request_fingerprint(r2, keep_fragments=True)) + r1 = Request("http://www.example.com") r2 = Request("http://www.example.com", method='POST') r3 = Request("http://www.example.com", method='POST', body=b'request body') From 229e722a03aced0fb62ec8c631b036c3e13e2188 Mon Sep 17 00:00:00 2001 From: Andrey Rahmatullin Date: Thu, 31 Oct 2019 14:46:02 +0500 Subject: [PATCH 066/113] Initial Python 2 removal (#4091) --- .travis.yml | 10 +----- README.rst | 2 +- docs/faq.rst | 4 +-- docs/intro/install.rst | 16 +++------- docs/topics/feed-exports.rst | 7 ++--- docs/topics/leaks.rst | 2 +- docs/topics/media-pipeline.rst | 2 +- docs/topics/request-response.rst | 4 +-- docs/topics/settings.rst | 5 +-- docs/topics/spiders.rst | 2 -- requirements-py2.txt | 18 ----------- scrapy/__init__.py | 4 +-- setup.py | 7 ++--- tests/requirements-py2.txt | 15 --------- tox.ini | 53 ++------------------------------ 15 files changed, 24 insertions(+), 127 deletions(-) delete mode 100644 requirements-py2.txt delete mode 100644 tests/requirements-py2.txt diff --git a/.travis.yml b/.travis.yml index 4c2498053..2352ef124 100644 --- a/.travis.yml +++ b/.travis.yml @@ -7,14 +7,6 @@ branches: - /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/ matrix: include: - - env: TOXENV=py27 - python: 2.7 - - env: TOXENV=py27-pinned - python: 2.7 - - env: TOXENV=py27-extra-deps - python: 2.7 - - env: TOXENV=pypy - python: 2.7 - env: TOXENV=pypy3 python: 3.5 - env: TOXENV=py35 @@ -70,4 +62,4 @@ deploy: on: tags: true repo: scrapy/scrapy - condition: "$TOXENV == py27 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" + condition: "$TOXENV == py37 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" diff --git a/README.rst b/README.rst index bd82bff06..fb4ca8e4f 100644 --- a/README.rst +++ b/README.rst @@ -40,7 +40,7 @@ https://scrapy.org Requirements ============ -* Python 2.7 or Python 3.5+ +* Python 3.5+ * Works on Linux, Windows, Mac OSX, BSD Install diff --git a/docs/faq.rst b/docs/faq.rst index 9733471bf..080d81981 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -69,11 +69,11 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars What Python versions does Scrapy support? ----------------------------------------- -Scrapy is supported under Python 2.7 and Python 3.5+ +Scrapy is supported under Python 3.5+ under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). -Python 2.6 support was dropped starting at Scrapy 0.20. Python 3 support was added in Scrapy 1.1. PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5. +Python 2 support was dropped in Scrapy 2.0. .. note:: For Python 3 support on Windows, it is recommended to use diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 51b41b4d7..e924b5303 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -7,7 +7,7 @@ Installation guide Installing Scrapy ================= -Scrapy runs on Python 2.7 and Python 3.5 or above +Scrapy runs on Python 3.5 or above under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). If you're using `Anaconda`_ or `Miniconda`_, you can install the package from @@ -102,10 +102,8 @@ just like any other Python package. (See :ref:`platform-specific guides ` below for non-Python dependencies that you may need to install beforehand). -Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. - -* If you want to install scrapy with Python 3, install scrapy within a Python 3 virtualenv. -* And if you want to install scrapy with Python 2, install scrapy within a Python 2 virtualenv. +Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy +only supports Python 3, make sure you created a Python 3 virtualenv. .. _virtualenv: https://virtualenv.pypa.io .. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/ @@ -149,16 +147,12 @@ typically too old and slow to catch up with latest Scrapy. To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install these dependencies:: - sudo apt-get install python-dev python-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev + sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev -- ``python-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev`` +- ``python3-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev`` are required for ``lxml`` - ``libssl-dev`` and ``libffi-dev`` are required for ``cryptography`` -If you want to install scrapy on Python 3, you’ll also need Python 3 development headers:: - - sudo apt-get install python3 python3-dev - Inside a :ref:`virtualenv `, you can install Scrapy with ``pip`` after that:: diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index af541db78..7481b1a99 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -99,12 +99,12 @@ The storages backends supported out of the box are: * :ref:`topics-feed-storage-fs` * :ref:`topics-feed-storage-ftp` - * :ref:`topics-feed-storage-s3` (requires botocore_ or boto_) + * :ref:`topics-feed-storage-s3` (requires botocore_) * :ref:`topics-feed-storage-stdout` Some storage backends may be unavailable if the required external libraries are not available. For example, the S3 backend is only available if the botocore_ -or boto_ library is installed (Scrapy supports boto_ only on Python 2). +library is installed. .. _topics-feed-uri-params: @@ -182,7 +182,7 @@ The feeds are stored on `Amazon S3`_. * ``s3://mybucket/path/to/export.csv`` * ``s3://aws_key:aws_secret@mybucket/path/to/export.csv`` - * Required external libraries: `botocore`_ (Python 2 and Python 3) or `boto`_ (Python 2 only) + * Required external libraries: `botocore`_ The AWS credentials can be passed as user/password in the URI, or they can be passed through the following settings: @@ -399,6 +399,5 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter .. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier .. _Amazon S3: https://aws.amazon.com/s3/ -.. _boto: https://github.com/boto/boto .. _botocore: https://github.com/boto/botocore .. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 8278e9849..793636f59 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -260,7 +260,7 @@ knowledge about Python internals. For more info about Guppy, refer to the Debugging memory leaks with muppy ================================= -If you're using Python 3, you can use muppy from `Pympler`_. +You can use muppy from `Pympler`_. .. _Pympler: https://pypi.org/project/Pympler/ diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 0ce431ff5..431cc6027 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -171,7 +171,7 @@ policy:: For more information, see `canned ACLs`_ in the Amazon S3 Developer Guide. -Because Scrapy uses ``boto`` / ``botocore`` internally you can also use other S3-like storages. Storages like +Because Scrapy uses ``botocore`` internally you can also use other S3-like storages. Storages like self-hosted `Minio`_ or `s3.scality`_. All you need to do is set endpoint option in you Scrapy settings:: AWS_ENDPOINT_URL = 'http://minio.example.com:9000' diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 727c67482..5a76e189e 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -596,8 +596,8 @@ Response objects (for single valued headers) or lists (for multi-valued headers). :type headers: dict - :param body: the response body. To access the decoded text as str (unicode - in Python 2) you can use ``response.text`` from an encoding-aware + :param body: the response body. To access the decoded text as str you can use + ``response.text`` from an encoding-aware :ref:`Response subclass `, such as :class:`TextResponse`. :type body: bytes diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 56375664f..a1d15a760 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -188,7 +188,6 @@ AWS_ENDPOINT_URL Default: ``None`` Endpoint URL used for S3-like storage, for example Minio or s3.scality. -Only supported with ``botocore`` library. .. setting:: AWS_USE_SSL @@ -199,7 +198,6 @@ Default: ``None`` Use this option if you want to disable SSL connection for communication with S3 or S3-like storage. By default SSL will be used. -Only supported with ``botocore`` library. .. setting:: AWS_VERIFY @@ -209,7 +207,7 @@ AWS_VERIFY Default: ``None`` Verify SSL connection between Scrapy and S3 or S3-like storage. By default -SSL verification will occur. Only supported with ``botocore`` library. +SSL verification will occur. .. setting:: AWS_REGION_NAME @@ -219,7 +217,6 @@ AWS_REGION_NAME Default: ``None`` The name of the region associated with the AWS client. -Only supported with ``botocore`` library. .. setting:: BOT_NAME diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index d60c93be6..d65a43afd 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -72,8 +72,6 @@ scrapy.Spider spider that crawls ``mywebsite.com`` would often be called ``mywebsite``. - .. note:: In Python 2 this must be ASCII only. - .. attribute:: allowed_domains An optional list of strings containing domains that this spider is diff --git a/requirements-py2.txt b/requirements-py2.txt deleted file mode 100644 index dde8d1c9c..000000000 --- a/requirements-py2.txt +++ /dev/null @@ -1,18 +0,0 @@ -parsel>=1.5.0 -PyDispatcher>=2.0.5 -w3lib>=1.17.0 -protego>=0.1.15 - -pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'" -queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'" -cryptography>=2.0 # Earlier versions would fail to install - -# Reference versions taken from -# https://packages.ubuntu.com/xenial/python/ -# https://packages.ubuntu.com/xenial/zope/ -cssselect>=0.9.1 -lxml>=3.5.0 -service_identity>=16.0.0 -six>=1.10.0 -Twisted>=16.0.0 -zope.interface>=4.1.3 diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 03ec6c667..230e5cee3 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -14,8 +14,8 @@ del pkgutil # Check minimum required Python version import sys -if sys.version_info < (2, 7): - print("Scrapy %s requires Python 2.7" % __version__) +if sys.version_info < (3, 5): + print("Scrapy %s requires Python 3.5" % __version__) sys.exit(1) # Ignore noisy twisted deprecation warnings diff --git a/setup.py b/setup.py index 2f5fca4c9..8f5f14f0d 100644 --- a/setup.py +++ b/setup.py @@ -50,8 +50,6 @@ setup( 'License :: OSI Approved :: BSD License', 'Operating System :: OS Independent', 'Programming Language :: Python', - 'Programming Language :: Python :: 2', - 'Programming Language :: Python :: 2.7', 'Programming Language :: Python :: 3', 'Programming Language :: Python :: 3.5', 'Programming Language :: Python :: 3.6', @@ -63,10 +61,9 @@ setup( 'Topic :: Software Development :: Libraries :: Application Frameworks', 'Topic :: Software Development :: Libraries :: Python Modules', ], - python_requires='>=2.7, !=3.0.*, !=3.1.*, !=3.2.*, !=3.3.*, !=3.4.*', + python_requires='>=3.5', install_requires=[ - 'Twisted>=16.0.0;python_version=="2.7"', - 'Twisted>=17.9.0;python_version>="3.5"', + 'Twisted>=17.9.0', 'cryptography>=2.0', 'cssselect>=0.9.1', 'lxml>=3.5.0', diff --git a/tests/requirements-py2.txt b/tests/requirements-py2.txt deleted file mode 100644 index f621eb4eb..000000000 --- a/tests/requirements-py2.txt +++ /dev/null @@ -1,15 +0,0 @@ -# Tests requirements -brotlipy -jmespath -mitmproxy==0.10.1 -mock -netlib==0.10.1 -pytest -pytest-cov -pytest-twisted -pytest-xdist -testfixtures - -# optional for shell wrapper tests -bpython -ipython<6.0 diff --git a/tox.ini b/tox.ini index fe925951b..821144381 100644 --- a/tox.ini +++ b/tox.ini @@ -4,18 +4,16 @@ # and then run "tox" from this directory. [tox] -envlist = py27 +envlist = py35 [testenv] deps = -ctests/constraints.txt - -rrequirements-py2.txt + -rrequirements-py3.txt + -rtests/requirements-py3.txt # Extras botocore>=1.3.23 - google-cloud-storage - leveldb Pillow>=3.4.2 - -rtests/requirements-py2.txt passenv = S3_TEST_FILE_URI AWS_ACCESS_KEY_ID @@ -25,42 +23,8 @@ passenv = commands = py.test --cov=scrapy --cov-report= {posargs:scrapy tests} -[testenv:py27-pinned] -basepython = python2.7 -deps = - -ctests/constraints.txt - cryptography==2.0 - cssselect==0.9.1 - lxml==3.5.0 - parsel==1.5.0 - Protego==0.1.15 - PyDispatcher==2.0.5 - pyOpenSSL==16.2.0 - queuelib==1.4.2 - service_identity==16.0.0 - six==1.10.0 - Twisted==16.0.0 - w3lib==1.17.0 - zope.interface==4.1.3 - -rtests/requirements-py2.txt - # Extras - botocore==1.3.23 - Pillow==3.4.2 - -[testenv:pypy] -basepython = pypy -commands = - py.test {posargs:scrapy tests} - [testenv:py35] basepython = python3.5 -deps = - -ctests/constraints.txt - -rrequirements-py3.txt - -rtests/requirements-py3.txt - # Extras - botocore>=1.3.23 - Pillow>=3.4.2 [testenv:py35-pinned] basepython = python3.5 @@ -86,19 +50,15 @@ deps = [testenv:py36] basepython = python3.6 -deps = {[testenv:py35]deps} [testenv:py37] basepython = python3.7 -deps = {[testenv:py35]deps} [testenv:py38] basepython = python3.8 -deps = {[testenv:py35]deps} [testenv:pypy3] basepython = pypy3 -deps = {[testenv:py35]deps} commands = py.test {posargs:scrapy tests} @@ -127,13 +87,6 @@ commands = [testenv:py38-extra-deps] basepython = python3.8 -deps = - {[testenv:py35]deps} - reppy - robotexclusionrulesparser - -[testenv:py27-extra-deps] -basepython = python2.7 deps = {[testenv]deps} reppy From 15c55d0c1d4a4384fd9725d47c305f52918e3831 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 31 Oct 2019 10:47:29 +0100 Subject: [PATCH 067/113] Remove LevelDB support (#4112) --- scrapy/extensions/httpcache.py | 71 -------------------- tests/requirements-py3.txt | 1 - tests/test_downloadermiddleware_httpcache.py | 9 --- 3 files changed, 81 deletions(-) diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 7c650a91e..f3fabf710 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -347,77 +347,6 @@ class FilesystemCacheStorage(object): return pickle.load(f) -class LeveldbCacheStorage(object): - - def __init__(self, settings): - warn("The LevelDB storage backend is deprecated.", - ScrapyDeprecationWarning, stacklevel=2) - import leveldb - self._leveldb = leveldb - self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True) - self.expiration_secs = settings.getint('HTTPCACHE_EXPIRATION_SECS') - self.db = None - - def open_spider(self, spider): - dbpath = os.path.join(self.cachedir, '%s.leveldb' % spider.name) - self.db = self._leveldb.LevelDB(dbpath) - - logger.debug("Using LevelDB cache storage in %(cachepath)s" % {'cachepath': dbpath}, extra={'spider': spider}) - - def close_spider(self, spider): - # Do compactation each time to save space and also recreate files to - # avoid them being removed in storages with timestamp-based autoremoval. - self.db.CompactRange() - del self.db - garbage_collect() - - def retrieve_response(self, spider, request): - data = self._read_data(spider, request) - if data is None: - return # not cached - url = data['url'] - status = data['status'] - headers = Headers(data['headers']) - body = data['body'] - respcls = responsetypes.from_args(headers=headers, url=url) - response = respcls(url=url, headers=headers, status=status, body=body) - return response - - def store_response(self, spider, request, response): - key = self._request_key(request) - data = { - 'status': response.status, - 'url': response.url, - 'headers': dict(response.headers), - 'body': response.body, - } - batch = self._leveldb.WriteBatch() - batch.Put(key + b'_data', pickle.dumps(data, protocol=2)) - batch.Put(key + b'_time', to_bytes(str(time()))) - self.db.Write(batch) - - def _read_data(self, spider, request): - key = self._request_key(request) - try: - ts = self.db.Get(key + b'_time') - except KeyError: - return # not found or invalid entry - - if 0 < self.expiration_secs < time() - float(ts): - return # expired - - try: - data = self.db.Get(key + b'_data') - except KeyError: - return # invalid entry - else: - return pickle.loads(data) - - def _request_key(self, request): - return to_bytes(request_fingerprint(request)) - - - def parse_cachecontrol(header): """Parse Cache-Control header diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index cb67bc40e..dd5b23cc3 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -1,6 +1,5 @@ # Tests requirements jmespath -leveldb; sys_platform != "win32" pytest pytest-cov pytest-twisted diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 972d400a4..950664ffe 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -156,15 +156,6 @@ class FilesystemStorageGzipTest(FilesystemStorageTest): return super(FilesystemStorageTest, self)._get_settings(**new_settings) -class LeveldbStorageTest(DefaultStorageTest): - - try: - pytest.importorskip('leveldb') - except SystemError: - pytestmark = pytest.mark.skip("Test module skipped - 'SystemError: bad call flags' occurs when >= Python 3.8") - storage_class = 'scrapy.extensions.httpcache.LeveldbCacheStorage' - - class DummyPolicyTest(_BaseTest): policy_class = 'scrapy.extensions.httpcache.DummyPolicy' From f02c3d1dcf3e4880388d19e961e7911be5dc54ff Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 31 Oct 2019 13:31:33 +0100 Subject: [PATCH 068/113] Use communicate() instead of wait() after killing the mock server (#4095) --- tests/mockserver.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/mockserver.py b/tests/mockserver.py index 77908284b..b766bb653 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -206,8 +206,7 @@ class MockServer(): def __exit__(self, exc_type, exc_value, traceback): self.proc.kill() - self.proc.wait() - time.sleep(0.2) + self.proc.communicate() def url(self, path, is_secure=False): host = self.http_address.replace('0.0.0.0', '127.0.0.1') From 439a3e59b8e858441f8d97dbc32f398db392330d Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Mon, 4 Nov 2019 10:35:58 -0300 Subject: [PATCH 069/113] Fix scrapy.utils.datatypes.LocalCache limit issue --- scrapy/utils/datatypes.py | 5 +++-- tests/test_utils_datatypes.py | 29 +++++++++++++++++++++++++++-- 2 files changed, 30 insertions(+), 4 deletions(-) diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index df2b99c28..f7e3240c1 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -315,8 +315,9 @@ class LocalCache(collections.OrderedDict): self.limit = limit def __setitem__(self, key, value): - while len(self) >= self.limit: - self.popitem(last=False) + if self.limit: + while len(self) >= self.limit: + self.popitem(last=False) super(LocalCache, self).__setitem__(key, value) diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 535095b8d..6ffd7c73c 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -7,7 +7,7 @@ if six.PY2: else: from collections.abc import Mapping, MutableMapping -from scrapy.utils.datatypes import CaselessDict, SequenceExclude +from scrapy.utils.datatypes import CaselessDict, SequenceExclude, LocalCache __doctests__ = ['scrapy.utils.datatypes'] @@ -242,6 +242,31 @@ class SequenceExcludeTest(unittest.TestCase): for v in [-3, "test", 1.1]: self.assertNotIn(v, d) + +class LocalCacheTest(unittest.TestCase): + + def test_cache_with_limit(self): + cache = LocalCache(limit=2) + cache['a'] = 1 + cache['b'] = 2 + cache['c'] = 3 + self.assertEqual(len(cache), 2) + self.assertNotIn('a', cache) + self.assertIn('b', cache) + self.assertIn('c', cache) + self.assertEqual(cache['b'], 2) + self.assertEqual(cache['c'], 3) + + def test_cache_without_limit(self): + max = 10**4 + cache = LocalCache() + for x in range(max): + cache[str(x)] = x + self.assertEqual(len(cache), max) + for x in range(max): + self.assertIn(str(x), cache) + self.assertEqual(cache[str(x)], x) + + if __name__ == "__main__": unittest.main() - From fed9fbe62d54175d70660498f2fba6b5b7f68a92 Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 4 Nov 2019 15:34:27 -0300 Subject: [PATCH 070/113] Update tests/test_utils_datatypes.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- tests/test_utils_datatypes.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 6ffd7c73c..fb2362829 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -7,7 +7,7 @@ if six.PY2: else: from collections.abc import Mapping, MutableMapping -from scrapy.utils.datatypes import CaselessDict, SequenceExclude, LocalCache +from scrapy.utils.datatypes import CaselessDict, LocalCache, SequenceExclude __doctests__ = ['scrapy.utils.datatypes'] From 613c66a034cebc885b90c8ba2a76aea2109ef72e Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 5 Nov 2019 09:45:51 -0300 Subject: [PATCH 071/113] Do not override built-in max function --- tests/test_utils_datatypes.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index fb2362829..7e671f627 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -258,12 +258,12 @@ class LocalCacheTest(unittest.TestCase): self.assertEqual(cache['c'], 3) def test_cache_without_limit(self): - max = 10**4 + maximum = 10**4 cache = LocalCache() - for x in range(max): + for x in range(maximum): cache[str(x)] = x - self.assertEqual(len(cache), max) - for x in range(max): + self.assertEqual(len(cache), maximum) + for x in range(maximum): self.assertIn(str(x), cache) self.assertEqual(cache[str(x)], x) From 698aa704b98e206cb755190f74d73a6ba47c5fac Mon Sep 17 00:00:00 2001 From: seregaxvm Date: Tue, 5 Nov 2019 18:30:01 +0300 Subject: [PATCH 072/113] Fix zsh completion file extension (#4122) --- extras/scrapy_zsh_completion | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion index 86c52c36c..e995947cb 100644 --- a/extras/scrapy_zsh_completion +++ b/extras/scrapy_zsh_completion @@ -61,7 +61,7 @@ _scrapy() { '-c[evaluate the code in the shell, print the result and exit]:code:(CODE)' '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' '--spider[use this spider]:spider:_scrapy_spiders' - '::file:_files -g \*.http' + '::file:_files -g \*.html' '::URL:_httpie_urls' ) _scrapy_glb_opts $options From 98caf055b5a343ff663f2c1150b301a3f4546a16 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 6 Nov 2019 11:53:46 +0100 Subject: [PATCH 073/113] =?UTF-8?q?Fix=20a=20typo:=20specifiy=20=E2=86=92?= =?UTF-8?q?=20specify=20(#4128)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scrapy/utils/misc.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index f638adb25..b74f34451 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -136,7 +136,7 @@ def create_instance(objcls, settings, crawler, *args, **kwargs): """ if settings is None: if crawler is None: - raise ValueError("Specifiy at least one of settings and crawler.") + raise ValueError("Specify at least one of settings and crawler.") settings = crawler.settings if crawler and hasattr(objcls, 'from_crawler'): return objcls.from_crawler(crawler, *args, **kwargs) From e8b1e46e85fbcdf22408d320f3cc61b6e802c5e2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Marc=20Hern=C3=A1ndez?= Date: Thu, 7 Nov 2019 14:05:01 +0100 Subject: [PATCH 074/113] Add pytest-flake8 (#3945) --- .travis.yml | 2 + conftest.py | 10 +++ pytest.ini | 253 ++++++++++++++++++++++++++++++++++++++++++++++++++++ tox.ini | 8 ++ 4 files changed, 273 insertions(+) diff --git a/.travis.yml b/.travis.yml index 2352ef124..f5b9d3f72 100644 --- a/.travis.yml +++ b/.travis.yml @@ -7,6 +7,8 @@ branches: - /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/ matrix: include: + - env: TOXENV=flake8 + python: 3.8 - env: TOXENV=pypy3 python: 3.5 - env: TOXENV=py35 diff --git a/conftest.py b/conftest.py index 06d65ba1d..d5d61ddd3 100644 --- a/conftest.py +++ b/conftest.py @@ -19,3 +19,13 @@ if six.PY3: def chdir(tmpdir): """Change to pytest-provided temporary directory""" tmpdir.chdir() + + +def pytest_collection_modifyitems(session, config, items): + # Avoid executing tests when executing `--flake8` flag (pytest-flake8) + try: + from pytest_flake8 import Flake8Item + if config.getoption('--flake8'): + items[:] = [item for item in items if isinstance(item, Flake8Item)] + except ImportError: + pass diff --git a/pytest.ini b/pytest.ini index 73d169601..0e4ee9ed1 100644 --- a/pytest.ini +++ b/pytest.ini @@ -4,3 +4,256 @@ python_files=test_*.py __init__.py python_classes= addopts = --doctest-modules --assert=plain twisted = 1 +flake8-ignore = + # extras + extras/qps-bench-server.py E261 E501 + extras/qpsclient.py E501 E261 E501 + # scrapy/commands + scrapy/commands/__init__.py E128 E501 + scrapy/commands/check.py F401 E501 W391 + scrapy/commands/crawl.py E501 + scrapy/commands/edit.py E501 + scrapy/commands/fetch.py E401 E302 E501 E128 E502 E731 + scrapy/commands/genspider.py E128 E501 E502 + scrapy/commands/list.py E302 + scrapy/commands/parse.py E128 E501 E731 E226 + scrapy/commands/runspider.py E501 + scrapy/commands/settings.py E302 E128 + scrapy/commands/shell.py E128 E501 E502 + scrapy/commands/startproject.py E502 E127 E501 E128 W391 + scrapy/commands/version.py E501 E128 W391 + scrapy/commands/view.py F401 E302 + # scrapy/contracts + scrapy/contracts/__init__.py E501 W504 + scrapy/contracts/default.py E502 E128 + # scrapy/core + scrapy/core/engine.py E261 E501 E128 E127 E306 E502 + scrapy/core/scheduler.py E501 + scrapy/core/scraper.py E501 E306 E261 E128 W391 W504 + scrapy/core/spidermw.py E501 E731 E502 E231 E126 E226 + scrapy/core/downloader/__init__.py F401 E501 + scrapy/core/downloader/contextfactory.py E501 E128 E126 + scrapy/core/downloader/middleware.py E501 E502 + scrapy/core/downloader/tls.py E501 E305 E241 + scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 W391 E126 E226 + scrapy/core/downloader/handlers/__init__.py E501 + scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 W391 + scrapy/core/downloader/handlers/http.py F401 + scrapy/core/downloader/handlers/http10.py E501 + scrapy/core/downloader/handlers/http11.py E501 + scrapy/core/downloader/handlers/s3.py E501 F401 E502 E128 E126 + # scrapy/downloadermiddlewares + scrapy/downloadermiddlewares/ajaxcrawl.py E302 E501 E226 + scrapy/downloadermiddlewares/decompression.py E501 + scrapy/downloadermiddlewares/defaultheaders.py E501 + scrapy/downloadermiddlewares/httpcache.py E501 E126 + scrapy/downloadermiddlewares/httpcompression.py E502 E128 + scrapy/downloadermiddlewares/httpproxy.py E501 + scrapy/downloadermiddlewares/redirect.py E501 W504 + scrapy/downloadermiddlewares/retry.py E501 E126 + scrapy/downloadermiddlewares/robotstxt.py F401 E501 + scrapy/downloadermiddlewares/stats.py E501 + # scrapy/extensions + scrapy/extensions/closespider.py E501 E502 E128 E123 + scrapy/extensions/corestats.py E302 E501 + scrapy/extensions/feedexport.py E128 E501 + scrapy/extensions/httpcache.py E128 E501 E303 F401 + scrapy/extensions/memdebug.py E501 + scrapy/extensions/spiderstate.py E302 E501 + scrapy/extensions/telnet.py E501 W504 + scrapy/extensions/throttle.py E501 + # scrapy/http + scrapy/http/__init__.py F401 + scrapy/http/common.py E501 + scrapy/http/cookies.py E501 + scrapy/http/headers.py W391 + scrapy/http/request/__init__.py E501 + scrapy/http/request/form.py E501 E123 + scrapy/http/request/json_request.py E501 + scrapy/http/response/__init__.py E501 E128 W293 W291 + scrapy/http/response/html.py E302 + scrapy/http/response/text.py E501 W293 E128 E124 + scrapy/http/response/xml.py E302 + # scrapy/linkextractors + scrapy/linkextractors/__init__.py E731 E502 E501 E402 F401 + scrapy/linkextractors/lxmlhtml.py E501 E731 E226 + # scrapy/loader + scrapy/loader/__init__.py E501 E502 E128 + scrapy/loader/common.py E302 + scrapy/loader/processors.py E501 + # scrapy/pipelines + scrapy/pipelines/__init__.py E302 + scrapy/pipelines/files.py E116 E501 E266 + scrapy/pipelines/images.py E265 E501 + scrapy/pipelines/media.py E125 E501 E266 + # scrapy/selector + scrapy/selector/__init__.py F403 F401 + scrapy/selector/unified.py F401 E501 E111 + # scrapy/settings + scrapy/settings/__init__.py E501 + scrapy/settings/default_settings.py E501 E261 E114 E116 E226 + scrapy/settings/deprecated.py E501 + # scrapy/spidermiddlewares + scrapy/spidermiddlewares/httperror.py E501 + scrapy/spidermiddlewares/offsite.py E501 + scrapy/spidermiddlewares/referer.py F401 E501 E129 W503 W504 + scrapy/spidermiddlewares/urllength.py E501 + # scrapy/spiders + scrapy/spiders/__init__.py F401 E501 E402 + scrapy/spiders/crawl.py E501 + scrapy/spiders/feed.py E501 E261 W391 + scrapy/spiders/init.py W391 + scrapy/spiders/sitemap.py E501 + # scrapy/utils + scrapy/utils/benchserver.py E501 + scrapy/utils/boto.py F401 + scrapy/utils/conf.py E402 E502 E501 + scrapy/utils/console.py E302 E261 F401 E306 E305 + scrapy/utils/curl.py F401 + scrapy/utils/datatypes.py E501 E226 + scrapy/utils/decorators.py E501 E302 + scrapy/utils/defer.py E501 E302 E128 + scrapy/utils/deprecate.py E128 E501 E127 E502 + scrapy/utils/display.py E302 + scrapy/utils/engine.py F401 E261 E302 + scrapy/utils/ftp.py E302 + scrapy/utils/gz.py E305 E501 E302 W504 + scrapy/utils/http.py F403 F401 W391 E226 + scrapy/utils/httpobj.py E302 E501 + scrapy/utils/iterators.py E501 E701 + scrapy/utils/job.py E302 + scrapy/utils/log.py E128 W503 + scrapy/utils/markup.py F403 F401 W292 + scrapy/utils/misc.py E501 E226 + scrapy/utils/multipart.py F403 F401 W292 + scrapy/utils/project.py E501 + scrapy/utils/python.py E501 E302 + scrapy/utils/reactor.py E302 E226 + scrapy/utils/reqser.py E501 + scrapy/utils/request.py E302 E127 E501 + scrapy/utils/response.py E501 E302 E128 + scrapy/utils/signal.py E501 E128 + scrapy/utils/sitemap.py E501 + scrapy/utils/spider.py E271 E302 E501 + scrapy/utils/ssl.py E501 + scrapy/utils/template.py E302 + scrapy/utils/test.py E302 E501 + scrapy/utils/url.py E501 F403 F401 E128 F405 + # scrapy + scrapy/__init__.py E402 E501 + scrapy/_monkeypatches.py W293 + scrapy/cmdline.py E502 E501 + scrapy/crawler.py E501 + scrapy/dupefilters.py E302 E501 E202 + scrapy/exceptions.py E302 E501 + scrapy/exporters.py E501 E261 E226 + scrapy/extension.py E302 + scrapy/interfaces.py E302 E501 W391 + scrapy/item.py E501 E128 + scrapy/link.py E501 W391 + scrapy/logformatter.py E501 W293 + scrapy/mail.py E402 E128 E501 E502 + scrapy/middleware.py E502 E128 E501 + scrapy/pqueues.py E501 + scrapy/resolver.py E302 + scrapy/responsetypes.py E128 E501 E305 + scrapy/robotstxt.py E302 E501 + scrapy/shell.py E501 + scrapy/signalmanager.py E501 + scrapy/spiderloader.py E225 F841 E501 E126 + scrapy/squeues.py E128 + scrapy/statscollectors.py E501 W391 + # tests + tests/__init__.py F401 E402 E501 + tests/mockserver.py E401 E501 E126 E123 F401 + tests/pipelines.py E302 F841 E226 + tests/spiders.py E302 E501 E127 + tests/test_closespider.py E501 E127 + tests/test_command_fetch.py E501 E261 + tests/test_command_parse.py F401 E302 E501 E128 E303 E226 + tests/test_command_shell.py E501 E128 + tests/test_commands.py F401 E128 E501 + tests/test_contracts.py E501 E128 W293 + tests/test_crawl.py E501 E741 E265 + tests/test_crawler.py F841 E306 E501 + tests/test_dependencies.py E302 F841 E501 E305 + tests/test_downloader_handlers.py E124 E127 E128 E225 E261 E265 F401 E501 E502 E701 E711 E126 E226 E123 + tests/test_downloadermiddleware.py E501 + tests/test_downloadermiddleware_ajaxcrawlable.py E302 E501 + tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126 + tests/test_downloadermiddleware_decompression.py E127 + tests/test_downloadermiddleware_defaultheaders.py E501 + tests/test_downloadermiddleware_downloadtimeout.py E501 + tests/test_downloadermiddleware_httpcache.py E713 E501 E302 E305 F401 + tests/test_downloadermiddleware_httpcompression.py E501 F401 E251 E126 E123 + tests/test_downloadermiddleware_httpproxy.py F401 E501 E128 + tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 + tests/test_downloadermiddleware_retry.py E501 E128 W293 E251 E502 E303 E126 + tests/test_downloadermiddleware_robotstxt.py E501 + tests/test_downloadermiddleware_stats.py E501 + tests/test_dupefilters.py E302 E221 E501 E741 W293 W291 E128 E124 + tests/test_engine.py E401 E501 E502 E128 E261 + tests/test_exporters.py E501 E731 E306 E128 E124 + tests/test_extension_telnet.py F401 F841 + tests/test_feedexport.py E501 F401 F841 E241 + tests/test_http_cookies.py E501 + tests/test_http_headers.py E302 E501 + tests/test_http_request.py F401 E402 E501 E231 E261 E127 E128 W293 E502 E128 E502 E126 E123 + tests/test_http_response.py E501 E301 E502 E128 E265 + tests/test_item.py E701 E128 E231 F841 E306 + tests/test_link.py E501 + tests/test_linkextractors.py E501 E128 E231 E124 + tests/test_loader.py E302 E501 E731 E303 E741 E128 E117 E241 + tests/test_logformatter.py E128 E501 E231 E122 E302 + tests/test_mail.py E302 E128 E501 E305 + tests/test_middleware.py E302 E501 E128 + tests/test_pipeline_crawl.py E131 E501 E128 E126 + tests/test_pipeline_files.py F401 E501 W293 E303 E272 E226 + tests/test_pipeline_images.py F401 F841 E501 E303 + tests/test_pipeline_media.py E501 E741 E731 E128 E261 E306 E502 + tests/test_request_cb_kwargs.py E501 + tests/test_responsetypes.py E501 E302 E305 + tests/test_robotstxt_interface.py F401 E302 E501 W291 E501 + tests/test_scheduler.py E501 E126 E123 + tests/test_selector.py F401 E501 E127 + tests/test_spider.py E501 F401 + tests/test_spidermiddleware.py E501 E226 + tests/test_spidermiddleware_depth.py W391 + tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 + tests/test_spidermiddleware_offsite.py E302 E501 E128 E111 W293 + tests/test_spidermiddleware_output_chain.py F401 E501 E302 W293 E226 + tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 W391 E241 E121 + tests/test_spidermiddleware_urllength.py W391 + tests/test_squeues.py E501 E302 E701 E741 + tests/test_utils_conf.py E501 E231 E303 E128 + tests/test_utils_console.py E302 E231 + tests/test_utils_curl.py E501 + tests/test_utils_datatypes.py E402 E501 E305 W391 + tests/test_utils_defer.py E306 E261 E501 E302 F841 E226 + tests/test_utils_deprecate.py F841 E306 E501 + tests/test_utils_http.py E302 E501 E502 E128 W391 W504 + tests/test_utils_httpobj.py E302 + tests/test_utils_iterators.py E501 E128 E129 E302 E303 E241 + tests/test_utils_log.py E741 E226 + tests/test_utils_python.py E501 E303 E731 E701 E305 + tests/test_utils_reqser.py F401 E501 E128 + tests/test_utils_request.py E302 E501 E128 E305 + tests/test_utils_response.py E501 + tests/test_utils_signal.py E741 F841 E302 E731 E226 + tests/test_utils_sitemap.py E302 E128 E501 E124 + tests/test_utils_spider.py E261 E302 E305 W391 + tests/test_utils_template.py E305 + tests/test_utils_url.py F401 E501 E127 E302 E305 E211 E125 E501 E226 E241 E126 E123 + tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 + tests/mocks/dummydbm.py E302 + tests/test_cmdline/__init__.py E502 E501 + tests/test_cmdline/extensions.py E302 W391 + tests/test_settings/__init__.py F401 E501 E128 + tests/test_settings/default_settings.py W391 + tests/test_spiderloader/__init__.py E128 E501 E302 + tests/test_spiderloader/test_spiders/spider0.py E302 + tests/test_spiderloader/test_spiders/spider1.py E302 + tests/test_spiderloader/test_spiders/spider2.py E302 + tests/test_spiderloader/test_spiders/spider3.py E302 + tests/test_spiderloader/test_spiders/nested/spider4.py E302 + tests/test_utils_misc/__init__.py E501 E231 diff --git a/tox.ini b/tox.ini index 821144381..cc3463a4d 100644 --- a/tox.ini +++ b/tox.ini @@ -62,6 +62,14 @@ basepython = pypy3 commands = py.test {posargs:scrapy tests} +[testenv:flake8] +basepython = python3.8 +deps = + {[testenv]deps} + pytest-flake8 +commands = + py.test --flake8 {posargs:scrapy tests} + [docs] changedir = docs deps = From c377c14e3263ce3c2bffa446cd0965006e845664 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Marc=20Hern=C3=A1ndez?= Date: Thu, 7 Nov 2019 17:47:35 +0100 Subject: [PATCH 075/113] Fix W391 Blank line at end of file (#4137) --- pytest.ini | 37 ++++++++++-------------- scrapy/commands/check.py | 1 - scrapy/commands/startproject.py | 1 - scrapy/commands/version.py | 1 - scrapy/core/downloader/handlers/ftp.py | 1 - scrapy/core/downloader/webclient.py | 1 - scrapy/core/scraper.py | 1 - scrapy/http/headers.py | 2 -- scrapy/interfaces.py | 1 - scrapy/link.py | 1 - scrapy/spiders/feed.py | 1 - scrapy/spiders/init.py | 1 - scrapy/statscollectors.py | 2 -- scrapy/utils/http.py | 1 - tests/test_cmdline/extensions.py | 1 - tests/test_settings/default_settings.py | 1 - tests/test_spidermiddleware_depth.py | 1 - tests/test_spidermiddleware_urllength.py | 1 - tests/test_utils_datatypes.py | 1 - tests/test_utils_http.py | 2 -- tests/test_utils_spider.py | 1 - 21 files changed, 16 insertions(+), 44 deletions(-) diff --git a/pytest.ini b/pytest.ini index 0e4ee9ed1..db5bee228 100644 --- a/pytest.ini +++ b/pytest.ini @@ -10,7 +10,7 @@ flake8-ignore = extras/qpsclient.py E501 E261 E501 # scrapy/commands scrapy/commands/__init__.py E128 E501 - scrapy/commands/check.py F401 E501 W391 + scrapy/commands/check.py F401 E501 scrapy/commands/crawl.py E501 scrapy/commands/edit.py E501 scrapy/commands/fetch.py E401 E302 E501 E128 E502 E731 @@ -20,8 +20,8 @@ flake8-ignore = scrapy/commands/runspider.py E501 scrapy/commands/settings.py E302 E128 scrapy/commands/shell.py E128 E501 E502 - scrapy/commands/startproject.py E502 E127 E501 E128 W391 - scrapy/commands/version.py E501 E128 W391 + scrapy/commands/startproject.py E502 E127 E501 E128 + scrapy/commands/version.py E501 E128 scrapy/commands/view.py F401 E302 # scrapy/contracts scrapy/contracts/__init__.py E501 W504 @@ -29,15 +29,15 @@ flake8-ignore = # scrapy/core scrapy/core/engine.py E261 E501 E128 E127 E306 E502 scrapy/core/scheduler.py E501 - scrapy/core/scraper.py E501 E306 E261 E128 W391 W504 + scrapy/core/scraper.py E501 E306 E261 E128 W504 scrapy/core/spidermw.py E501 E731 E502 E231 E126 E226 scrapy/core/downloader/__init__.py F401 E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 scrapy/core/downloader/tls.py E501 E305 E241 - scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 W391 E126 E226 + scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 E126 E226 scrapy/core/downloader/handlers/__init__.py E501 - scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 W391 + scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 scrapy/core/downloader/handlers/http.py F401 scrapy/core/downloader/handlers/http10.py E501 scrapy/core/downloader/handlers/http11.py E501 @@ -66,7 +66,6 @@ flake8-ignore = scrapy/http/__init__.py F401 scrapy/http/common.py E501 scrapy/http/cookies.py E501 - scrapy/http/headers.py W391 scrapy/http/request/__init__.py E501 scrapy/http/request/form.py E501 E123 scrapy/http/request/json_request.py E501 @@ -101,8 +100,7 @@ flake8-ignore = # scrapy/spiders scrapy/spiders/__init__.py F401 E501 E402 scrapy/spiders/crawl.py E501 - scrapy/spiders/feed.py E501 E261 W391 - scrapy/spiders/init.py W391 + scrapy/spiders/feed.py E501 E261 scrapy/spiders/sitemap.py E501 # scrapy/utils scrapy/utils/benchserver.py E501 @@ -118,7 +116,7 @@ flake8-ignore = scrapy/utils/engine.py F401 E261 E302 scrapy/utils/ftp.py E302 scrapy/utils/gz.py E305 E501 E302 W504 - scrapy/utils/http.py F403 F401 W391 E226 + scrapy/utils/http.py F403 F401 E226 scrapy/utils/httpobj.py E302 E501 scrapy/utils/iterators.py E501 E701 scrapy/utils/job.py E302 @@ -148,9 +146,9 @@ flake8-ignore = scrapy/exceptions.py E302 E501 scrapy/exporters.py E501 E261 E226 scrapy/extension.py E302 - scrapy/interfaces.py E302 E501 W391 + scrapy/interfaces.py E302 E501 scrapy/item.py E501 E128 - scrapy/link.py E501 W391 + scrapy/link.py E501 scrapy/logformatter.py E501 W293 scrapy/mail.py E402 E128 E501 E502 scrapy/middleware.py E502 E128 E501 @@ -162,7 +160,7 @@ flake8-ignore = scrapy/signalmanager.py E501 scrapy/spiderloader.py E225 F841 E501 E126 scrapy/squeues.py E128 - scrapy/statscollectors.py E501 W391 + scrapy/statscollectors.py E501 # tests tests/__init__.py F401 E402 E501 tests/mockserver.py E401 E501 E126 E123 F401 @@ -218,20 +216,18 @@ flake8-ignore = tests/test_selector.py F401 E501 E127 tests/test_spider.py E501 F401 tests/test_spidermiddleware.py E501 E226 - tests/test_spidermiddleware_depth.py W391 tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 tests/test_spidermiddleware_offsite.py E302 E501 E128 E111 W293 tests/test_spidermiddleware_output_chain.py F401 E501 E302 W293 E226 - tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 W391 E241 E121 - tests/test_spidermiddleware_urllength.py W391 + tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 E241 E121 tests/test_squeues.py E501 E302 E701 E741 tests/test_utils_conf.py E501 E231 E303 E128 tests/test_utils_console.py E302 E231 tests/test_utils_curl.py E501 - tests/test_utils_datatypes.py E402 E501 E305 W391 + tests/test_utils_datatypes.py E402 E501 E305 tests/test_utils_defer.py E306 E261 E501 E302 F841 E226 tests/test_utils_deprecate.py F841 E306 E501 - tests/test_utils_http.py E302 E501 E502 E128 W391 W504 + tests/test_utils_http.py E302 E501 E502 E128 W504 tests/test_utils_httpobj.py E302 tests/test_utils_iterators.py E501 E128 E129 E302 E303 E241 tests/test_utils_log.py E741 E226 @@ -241,15 +237,14 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 E302 E731 E226 tests/test_utils_sitemap.py E302 E128 E501 E124 - tests/test_utils_spider.py E261 E302 E305 W391 + tests/test_utils_spider.py E261 E302 E305 tests/test_utils_template.py E305 tests/test_utils_url.py F401 E501 E127 E302 E305 E211 E125 E501 E226 E241 E126 E123 tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 tests/mocks/dummydbm.py E302 tests/test_cmdline/__init__.py E502 E501 - tests/test_cmdline/extensions.py E302 W391 + tests/test_cmdline/extensions.py E302 tests/test_settings/__init__.py F401 E501 E128 - tests/test_settings/default_settings.py W391 tests/test_spiderloader/__init__.py E128 E501 E302 tests/test_spiderloader/test_spiders/spider0.py E302 tests/test_spiderloader/test_spiders/spider1.py E302 diff --git a/scrapy/commands/check.py b/scrapy/commands/check.py index ab73e85e7..3e6c11b7d 100644 --- a/scrapy/commands/check.py +++ b/scrapy/commands/check.py @@ -96,4 +96,3 @@ class Command(ScrapyCommand): result.printErrors() result.printSummary(start, stop) self.exitcode = int(not result.wasSuccessful()) - diff --git a/scrapy/commands/startproject.py b/scrapy/commands/startproject.py index 67337c26e..3b9f6eabb 100644 --- a/scrapy/commands/startproject.py +++ b/scrapy/commands/startproject.py @@ -119,4 +119,3 @@ class Command(ScrapyCommand): _templates_base_dir = self.settings['TEMPLATES_DIR'] or \ join(scrapy.__path__[0], 'templates') return join(_templates_base_dir, 'project') - diff --git a/scrapy/commands/version.py b/scrapy/commands/version.py index 577365c3b..8651948f7 100644 --- a/scrapy/commands/version.py +++ b/scrapy/commands/version.py @@ -30,4 +30,3 @@ class Command(ScrapyCommand): print(patt % (name, version)) else: print("Scrapy %s" % scrapy.__version__) - diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 806a537d4..39ed67a1a 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -112,4 +112,3 @@ class FTPDownloadHandler(object): httpcode = self.CODE_MAPPING.get(ftpcode, self.CODE_MAPPING["default"]) return Response(url=request.url, status=httpcode, body=to_bytes(message)) raise result.type(result.value) - diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index 3a5890ed0..3fe13414a 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -157,4 +157,3 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): def gotHeaders(self, headers): self.headers_time = time() self.response_headers = headers - diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 1f389cf2e..40de6b87a 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -244,4 +244,3 @@ class Scraper(object): return self.signals.send_catch_log_deferred( signal=signals.item_scraped, item=output, response=response, spider=spider) - diff --git a/scrapy/http/headers.py b/scrapy/http/headers.py index 62507eb19..f3b46b994 100644 --- a/scrapy/http/headers.py +++ b/scrapy/http/headers.py @@ -91,5 +91,3 @@ class Headers(CaselessDict): def __copy__(self): return self.__class__(self) copy = __copy__ - - diff --git a/scrapy/interfaces.py b/scrapy/interfaces.py index 89ad2b14f..d48babc3c 100644 --- a/scrapy/interfaces.py +++ b/scrapy/interfaces.py @@ -15,4 +15,3 @@ class ISpiderLoader(Interface): def find_by_request(request): """Return the list of spiders names that can handle the given request""" - diff --git a/scrapy/link.py b/scrapy/link.py index 2c8301680..f0638ced2 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -39,4 +39,3 @@ class Link(object): def __repr__(self): return 'Link(url=%r, text=%r, fragment=%r, nofollow=%r)' % \ (self.url, self.text, self.fragment, self.nofollow) - diff --git a/scrapy/spiders/feed.py b/scrapy/spiders/feed.py index 06e212e1c..197812a26 100644 --- a/scrapy/spiders/feed.py +++ b/scrapy/spiders/feed.py @@ -133,4 +133,3 @@ class CSVFeedSpider(Spider): raise NotConfigured('You must define parse_row method in order to scrape this CSV feed') response = self.adapt_response(response) return self.parse_rows(response) - diff --git a/scrapy/spiders/init.py b/scrapy/spiders/init.py index 2efb1a869..fd41133ea 100644 --- a/scrapy/spiders/init.py +++ b/scrapy/spiders/init.py @@ -29,4 +29,3 @@ class InitSpider(Spider): spider """ return self.initialized() - diff --git a/scrapy/statscollectors.py b/scrapy/statscollectors.py index 6da9ddcd2..f0bfaed34 100644 --- a/scrapy/statscollectors.py +++ b/scrapy/statscollectors.py @@ -80,5 +80,3 @@ class DummyStatsCollector(StatsCollector): def min_value(self, key, value, spider=None): pass - - diff --git a/scrapy/utils/http.py b/scrapy/utils/http.py index b6e05c862..ad49ef3e9 100644 --- a/scrapy/utils/http.py +++ b/scrapy/utils/http.py @@ -34,4 +34,3 @@ def decode_chunked_transfer(chunked_body): body += t[:size] t = t[size+2:] return body - diff --git a/tests/test_cmdline/extensions.py b/tests/test_cmdline/extensions.py index 72867eb56..28456b55d 100644 --- a/tests/test_cmdline/extensions.py +++ b/tests/test_cmdline/extensions.py @@ -12,4 +12,3 @@ class TestExtension(object): class DummyExtension(object): pass - diff --git a/tests/test_settings/default_settings.py b/tests/test_settings/default_settings.py index c24b5a9b9..26a555275 100644 --- a/tests/test_settings/default_settings.py +++ b/tests/test_settings/default_settings.py @@ -2,4 +2,3 @@ TEST_DEFAULT = 'defvalue' TEST_DICT = {'key': 'val'} - diff --git a/tests/test_spidermiddleware_depth.py b/tests/test_spidermiddleware_depth.py index 3685d5a6f..71cca2472 100644 --- a/tests/test_spidermiddleware_depth.py +++ b/tests/test_spidermiddleware_depth.py @@ -40,4 +40,3 @@ class TestDepthMiddleware(TestCase): def tearDown(self): self.stats.close_spider(self.spider, '') - diff --git a/tests/test_spidermiddleware_urllength.py b/tests/test_spidermiddleware_urllength.py index a0aae0fdd..5ef2b23fd 100644 --- a/tests/test_spidermiddleware_urllength.py +++ b/tests/test_spidermiddleware_urllength.py @@ -18,4 +18,3 @@ class TestUrlLengthMiddleware(TestCase): spider = Spider('foo') out = list(mw.process_spider_output(res, reqs, spider)) self.assertEqual(out, [short_url_req]) - diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 535095b8d..9455172fa 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -244,4 +244,3 @@ class SequenceExcludeTest(unittest.TestCase): if __name__ == "__main__": unittest.main() - diff --git a/tests/test_utils_http.py b/tests/test_utils_http.py index 583105673..2524153ea 100644 --- a/tests/test_utils_http.py +++ b/tests/test_utils_http.py @@ -16,5 +16,3 @@ class ChunkedTest(unittest.TestCase): "This is the data in the first chunk\r\n" + "and this is the second one\r\n" + "consequence") - - diff --git a/tests/test_utils_spider.py b/tests/test_utils_spider.py index 045e72117..d9de1ce77 100644 --- a/tests/test_utils_spider.py +++ b/tests/test_utils_spider.py @@ -34,4 +34,3 @@ class UtilsSpidersTestCase(unittest.TestCase): if __name__ == "__main__": unittest.main() - From d874c4d90bcf96c7e5b507babaa2a45a233da506 Mon Sep 17 00:00:00 2001 From: Andrey Rahmatullin Date: Thu, 7 Nov 2019 22:02:17 +0500 Subject: [PATCH 076/113] Remove the old Python 2 PyPy installation code from .travis.yml (#4138) --- .travis.yml | 7 ------- 1 file changed, 7 deletions(-) diff --git a/.travis.yml b/.travis.yml index f5b9d3f72..0e77af9fd 100644 --- a/.travis.yml +++ b/.travis.yml @@ -27,13 +27,6 @@ matrix: python: 3.6 install: - | - if [ "$TOXENV" = "pypy" ]; then - export PYPY_VERSION="pypy-6.0.0-linux_x86_64-portable" - wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2" - tar -jxf ${PYPY_VERSION}.tar.bz2 - virtualenv --python="$PYPY_VERSION/bin/pypy" "$HOME/virtualenvs/$PYPY_VERSION" - source "$HOME/virtualenvs/$PYPY_VERSION/bin/activate" - fi if [ "$TOXENV" = "pypy3" ]; then export PYPY_VERSION="pypy3.5-5.9-beta-linux_x86_64-portable" wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2" From aef98188facfc79dc574d8a86200b4e95b96b880 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 7 Nov 2019 18:06:55 +0100 Subject: [PATCH 077/113] Improve the details about request serialization requirements for JOBDIR --- docs/topics/jobs.rst | 31 ++++--------------------------- 1 file changed, 4 insertions(+), 27 deletions(-) diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst index 9fd311c69..f5542495b 100644 --- a/docs/topics/jobs.rst +++ b/docs/topics/jobs.rst @@ -71,34 +71,11 @@ on cookies. Request serialization --------------------- -Requests must be serializable by the ``pickle`` module, in order for persistence -to work, so you should make sure that your requests are serializable. - -The most common issue here is to use ``lambda`` functions on request callbacks that -can't be persisted. - -So, for example, this won't work:: - - def some_callback(self, response): - somearg = 'test' - return scrapy.Request('http://www.example.com', - callback=lambda r: self.other_callback(r, somearg)) - - def other_callback(self, response, somearg): - print("the argument passed is: %s" % somearg) - -But this will:: - - def some_callback(self, response): - somearg = 'test' - return scrapy.Request('http://www.example.com', - callback=self.other_callback, cb_kwargs={'somearg': somearg}) - - def other_callback(self, response, somearg): - print("the argument passed is: %s" % somearg) +For persistence to work, :class:`~scrapy.http.Request` objects must be +serializable with :mod:`pickle`, except for the ``callback`` and ``errback`` +values passed to their ``__init__`` method, which must be methods of the +runnning :class:`~scrapy.spiders.Spider` class. If you wish to log the requests that couldn't be serialized, you can set the :setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page. It is ``False`` by default. - -.. _pickle: https://docs.python.org/library/pickle.html From 1df5755699eac5a98239ae73dfb82908706bf03b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Fri, 8 Nov 2019 16:00:10 +0100 Subject: [PATCH 078/113] Set the bases for testing code examples from the documentation --- docs/conftest.py | 16 ++++++++++++++++ pytest.ini | 20 +++++++++++++++++++- tests/requirements-py3.txt | 1 + tox.ini | 6 +++--- 4 files changed, 39 insertions(+), 4 deletions(-) create mode 100644 docs/conftest.py diff --git a/docs/conftest.py b/docs/conftest.py new file mode 100644 index 000000000..91c1d4428 --- /dev/null +++ b/docs/conftest.py @@ -0,0 +1,16 @@ +from doctest import ELLIPSIS + +from sybil import Sybil +from sybil.parsers.codeblock import CodeBlockParser +from sybil.parsers.doctest import DocTestParser +from sybil.parsers.skip import skip + + +pytest_collect_file = Sybil( + parsers=[ + DocTestParser(optionflags=ELLIPSIS), + CodeBlockParser(future_imports=['print_function']), + skip, + ], + pattern='*.rst', +).pytest() diff --git a/pytest.ini b/pytest.ini index db5bee228..8c5a2cd54 100644 --- a/pytest.ini +++ b/pytest.ini @@ -2,7 +2,25 @@ usefixtures = chdir python_files=test_*.py __init__.py python_classes= -addopts = --doctest-modules --assert=plain +addopts = + --assert=plain + --doctest-modules + --ignore=docs/_ext + --ignore=docs/conf.py + --ignore=docs/intro/tutorial.rst + --ignore=docs/news.rst + --ignore=docs/topics/commands.rst + --ignore=docs/topics/debug.rst + --ignore=docs/topics/developer-tools.rst + --ignore=docs/topics/dynamic-content.rst + --ignore=docs/topics/items.rst + --ignore=docs/topics/leaks.rst + --ignore=docs/topics/loaders.rst + --ignore=docs/topics/selectors.rst + --ignore=docs/topics/shell.rst + --ignore=docs/topics/stats.rst + --ignore=docs/topics/telnetconsole.rst + --ignore=docs/utils twisted = 1 flake8-ignore = # extras diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index dd5b23cc3..2e8d319d2 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -4,6 +4,7 @@ pytest pytest-cov pytest-twisted pytest-xdist +sybil testfixtures # optional for shell wrapper tests diff --git a/tox.ini b/tox.ini index cc3463a4d..3668058c3 100644 --- a/tox.ini +++ b/tox.ini @@ -21,7 +21,7 @@ passenv = GCS_TEST_FILE_URI GCS_PROJECT_ID commands = - py.test --cov=scrapy --cov-report= {posargs:scrapy tests} + py.test --cov=scrapy --cov-report= {posargs:docs scrapy tests} [testenv:py35] basepython = python3.5 @@ -60,7 +60,7 @@ basepython = python3.8 [testenv:pypy3] basepython = pypy3 commands = - py.test {posargs:scrapy tests} + py.test {posargs:docs scrapy tests} [testenv:flake8] basepython = python3.8 @@ -68,7 +68,7 @@ deps = {[testenv]deps} pytest-flake8 commands = - py.test --flake8 {posargs:scrapy tests} + py.test --flake8 {posargs:docs scrapy tests} [docs] changedir = docs From 084a1cda6dfd94a3671d49c362db2ee6ea88a10d Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 15:41:00 +0530 Subject: [PATCH 079/113] Adding test --- tests/test_feedexport.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 8c0e5cd3d..16916f728 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -415,7 +415,7 @@ class FeedExportTest(unittest.TestCase): spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(res_path, 'rb') as f: + with open(res_uri, 'rb') as f: content = f.read() finally: @@ -850,12 +850,12 @@ class FeedExportTest(unittest.TestCase): def test_pathlib_uri(self): tmpdir = tempfile.mkdtemp() feed_uri = Path(tmpdir) / 'res' - feed_uri=str(feed_uri) res_uri = urljoin('file:', pathname2url(feed_uri)) settings = { 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, - 'FEED_URI': res_uri, + 'FEED_URI': feed_uri, + 'FEED_URI_ISPATH' : True } data = yield self.exported_no_data(settings) From 0042c389eb2d44d017bc8af069c7dd7ebd9319bd Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 15:57:58 +0530 Subject: [PATCH 080/113] Adding test --- tests/test_feedexport.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 16916f728..275579959 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -855,7 +855,6 @@ class FeedExportTest(unittest.TestCase): 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, 'FEED_URI': feed_uri, - 'FEED_URI_ISPATH' : True } data = yield self.exported_no_data(settings) From 9e6e2dde2b7736278075d6a8268511a3bc44b8b5 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 16:10:37 +0530 Subject: [PATCH 081/113] Adding test --- tests/test_feedexport.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 275579959..b6e4a5449 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -415,7 +415,7 @@ class FeedExportTest(unittest.TestCase): spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(res_uri, 'rb') as f: + with open(defaults['FEED_URI'], 'rb') as f: content = f.read() finally: From 970c3be1603483a61637b94afdb965eb24342744 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 18:34:15 +0530 Subject: [PATCH 082/113] Added Test --- scrapy/extensions/feedexport.py | 2 +- tests/test_feedexport.py | 7 ++++--- 2 files changed, 5 insertions(+), 4 deletions(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 981efee55..8dacce459 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -201,7 +201,7 @@ class FeedExporter(object): self.settings = settings if not settings['FEED_URI']: raise NotConfigured - self.urifmt=str(settings['FEED_URI']) + self.urifmt = str(settings['FEED_URI']) self.format = settings['FEED_FORMAT'].lower() self.export_encoding = settings['FEED_EXPORT_ENCODING'] self.storages = self._load_components('FEED_STORAGES') diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index b6e4a5449..11d32bd14 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -407,6 +407,7 @@ class FeedExportTest(unittest.TestCase): defaults = { 'FEED_URI': res_uri, 'FEED_FORMAT': 'csv', + 'FEED_PATH': res_path } defaults.update(settings or {}) try: @@ -415,7 +416,7 @@ class FeedExportTest(unittest.TestCase): spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(defaults['FEED_URI'], 'rb') as f: + with open(defaults['FEED_PATH'], 'rb') as f: content = f.read() finally: @@ -855,8 +856,8 @@ class FeedExportTest(unittest.TestCase): 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, 'FEED_URI': feed_uri, + 'FEED_PATH': feed_uri } - data = yield self.exported_no_data(settings) self.assertEqual(data, b'') - shutil.rmtree(tmpdir, ignore_errors=True) \ No newline at end of file + shutil.rmtree(tmpdir, ignore_errors=True) From 0c2dcd5092eccf08399f0838d86d598329cc3a28 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 18:35:50 +0530 Subject: [PATCH 083/113] Added Test --- requirements-py2.txt | 19 ------------------- 1 file changed, 19 deletions(-) delete mode 100644 requirements-py2.txt diff --git a/requirements-py2.txt b/requirements-py2.txt deleted file mode 100644 index 42e057417..000000000 --- a/requirements-py2.txt +++ /dev/null @@ -1,19 +0,0 @@ -parsel>=1.5.0 -PyDispatcher>=2.0.5 -w3lib>=1.17.0 -protego>=0.1.15 - -pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'" -queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'" -cryptography>=2.0 # Earlier versions would fail to install - -# Reference versions taken from -# https://packages.ubuntu.com/xenial/python/ -# https://packages.ubuntu.com/xenial/zope/ -cssselect>=0.9.1 -lxml>=3.5.0 -service_identity>=16.0.0 -six>=1.10.0 -Twisted>=16.0.0 -zope.interface>=4.1.3 -pathlib2>=2.0 From f39ff4945854fb5d98389fabfa2d2e5a059c0643 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 18:54:21 +0530 Subject: [PATCH 084/113] Added Test --- tests/test_feedexport.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 11d32bd14..9b07c2051 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -851,7 +851,6 @@ class FeedExportTest(unittest.TestCase): def test_pathlib_uri(self): tmpdir = tempfile.mkdtemp() feed_uri = Path(tmpdir) / 'res' - res_uri = urljoin('file:', pathname2url(feed_uri)) settings = { 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, From 50eaabe1fc540218f9f04197810367154eb3e102 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Mon, 11 Nov 2019 20:00:26 +0530 Subject: [PATCH 085/113] Added Test --- tests/test_feedexport.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 9b07c2051..2819f8f0b 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -416,7 +416,7 @@ class FeedExportTest(unittest.TestCase): spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(defaults['FEED_PATH'], 'rb') as f: + with open(str(defaults['FEED_PATH']), 'rb') as f: content = f.read() finally: From 79d2f99995a12ffc19ab7cd2fda10112cf5e9b65 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 12 Nov 2019 08:08:50 +0100 Subject: [PATCH 086/113] Make tutorial doctests pass --- docs/_tests/quotes1.html | 281 +++++++++++++++++++++++++++++++++++++++ docs/conftest.py | 17 ++- docs/intro/tutorial.rst | 23 ++-- pytest.ini | 1 - 4 files changed, 310 insertions(+), 12 deletions(-) create mode 100644 docs/_tests/quotes1.html diff --git a/docs/_tests/quotes1.html b/docs/_tests/quotes1.html new file mode 100644 index 000000000..71aff8847 --- /dev/null +++ b/docs/_tests/quotes1.html @@ -0,0 +1,281 @@ + + + + + Quotes to Scrape + + + + +
+
+ +
+

+ + Login + +

+
+
+ + +
+
+ +
+ “The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.” + by + (about) + +
+ Tags: + + + change + + deep-thoughts + + thinking + + world + +
+
+ +
+ “It is our choices, Harry, that show what we truly are, far more than our abilities.” + by + (about) + +
+ Tags: + + + abilities + + choices + +
+
+ +
+ “There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.” + by + (about) + +
+ Tags: + + + inspirational + + life + + live + + miracle + + miracles + +
+
+ +
+ “The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.” + by + (about) + +
+ Tags: + + + aliteracy + + books + + classic + + humor + +
+
+ +
+ “Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.” + by + (about) + +
+ Tags: + + + be-yourself + + inspirational + +
+
+ +
+ “Try not to become a man of success. Rather become a man of value.” + by + (about) + +
+ Tags: + + + adulthood + + success + + value + +
+
+ +
+ “It is better to be hated for what you are than to be loved for what you are not.” + by + (about) + +
+ Tags: + + + life + + love + +
+
+ +
+ “I have not failed. I've just found 10,000 ways that won't work.” + by + (about) + +
+ Tags: + + + edison + + failure + + inspirational + + paraphrased + +
+
+ +
+ “A woman is like a tea bag; you never know how strong it is until it's in hot water.” + by + (about) + + +
+ +
+ “A day without sunshine is like, you know, night.” + by + (about) + +
+ Tags: + + + humor + + obvious + + simile + +
+
+ + +
+
+ +

Top Ten tags

+ + + love + + + + inspirational + + + + life + + + + humor + + + + books + + + + reading + + + + friendship + + + + friends + + + + truth + + + + simile + + + +
+
+ +
+ + + \ No newline at end of file diff --git a/docs/conftest.py b/docs/conftest.py index 91c1d4428..8c735e838 100644 --- a/docs/conftest.py +++ b/docs/conftest.py @@ -1,16 +1,29 @@ -from doctest import ELLIPSIS +import os +from doctest import ELLIPSIS, NORMALIZE_WHITESPACE +from scrapy.http.response.html import HtmlResponse from sybil import Sybil from sybil.parsers.codeblock import CodeBlockParser from sybil.parsers.doctest import DocTestParser from sybil.parsers.skip import skip +def load_response(url, filename): + input_path = os.path.join(os.path.dirname(__file__), '_tests', filename) + with open(input_path, 'rb') as input_file: + return HtmlResponse(url, body=input_file.read()) + + +def setup(namespace): + namespace['load_response'] = load_response + + pytest_collect_file = Sybil( parsers=[ - DocTestParser(optionflags=ELLIPSIS), + DocTestParser(optionflags=ELLIPSIS | NORMALIZE_WHITESPACE), CodeBlockParser(future_imports=['print_function']), skip, ], pattern='*.rst', + setup=setup, ).pytest() diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 0629b9e19..996e3b475 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -235,13 +235,16 @@ You will see something like:: [s] shelp() Shell help (print this help) [s] fetch(req_or_url) Fetch request (or URL) and update local objects [s] view(response) View response in a browser - >>> Using the shell, you can try selecting elements using `CSS`_ with the response -object:: +object: - >>> response.css('title') - [] +.. invisible-code-block: python + + response = load_response('http://quotes.toscrape.com/page/1/', 'quotes1.html') + +>>> response.css('title') +[] The result of running ``response.css('title')`` is a list-like object called :class:`~scrapy.selector.SelectorList`, which represents a list of @@ -372,6 +375,9 @@ we want:: We get a list of selectors for the quote HTML elements with:: >>> response.css("div.quote") + [, + , + ...] Each of the selectors returned by the query above allows us to run further queries over their sub-elements. Let's assign the first selector to a @@ -404,10 +410,9 @@ quotes elements and put them together into a Python dictionary:: ... author = quote.css("small.author::text").get() ... tags = quote.css("div.tags a.tag::text").getall() ... print(dict(text=text, author=author, tags=tags)) - {'tags': ['change', 'deep-thoughts', 'thinking', 'world'], 'author': 'Albert Einstein', 'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'} - {'tags': ['abilities', 'choices'], 'author': 'J.K. Rowling', 'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”'} - ... a few more of these, omitted for brevity - >>> + {'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']} + {'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']} + ... Extracting data in our spider ----------------------------- @@ -521,7 +526,7 @@ There is also an ``attrib`` property available (see :ref:`selecting-attributes` for more):: >>> response.css('li.next a').attrib['href'] - '/page/2' + '/page/2/' Let's see now our spider modified to recursively follow the link to the next page, extracting data from it:: diff --git a/pytest.ini b/pytest.ini index 8c5a2cd54..3f1cc5800 100644 --- a/pytest.ini +++ b/pytest.ini @@ -7,7 +7,6 @@ addopts = --doctest-modules --ignore=docs/_ext --ignore=docs/conf.py - --ignore=docs/intro/tutorial.rst --ignore=docs/news.rst --ignore=docs/topics/commands.rst --ignore=docs/topics/debug.rst From 7b7bb028f45a06b3444950521ea582d0e83691ba Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 12 Nov 2019 08:49:06 +0100 Subject: [PATCH 087/113] Use intersphinx for links to the Sphinx documentation --- docs/conf.py | 1 + docs/contributing.rst | 17 ++++++++--------- 2 files changed, 9 insertions(+), 9 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 34dd5bcb7..0cc1dc22b 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -273,4 +273,5 @@ coverage_ignore_pyobjects = [ intersphinx_mapping = { 'python': ('https://docs.python.org/3', None), + 'sphinx': ('https://www.sphinx-doc.org/en/stable', None), } diff --git a/docs/contributing.rst b/docs/contributing.rst index 28dea74de..f084bd23d 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -177,20 +177,19 @@ Documentation policies ====================== For reference documentation of API members (classes, methods, etc.) use -docstrings and make sure that the Sphinx documentation uses the autodoc_ -extension to pull the docstrings. API reference documentation should follow -docstring conventions (`PEP 257`_) and be IDE-friendly: short, to the point, -and it may provide short examples. +docstrings and make sure that the Sphinx documentation uses the +:mod:`~sphinx.ext.autodoc` extension to pull the docstrings. API reference +documentation should follow docstring conventions (`PEP 257`_) and be +IDE-friendly: short, to the point, and it may provide short examples. Other types of documentation, such as tutorials or topics, should be covered in files within the ``docs/`` directory. This includes documentation that is specific to an API member, but goes beyond API reference documentation. -In any case, if something is covered in a docstring, use the autodoc_ -extension to pull the docstring into the documentation instead of duplicating -the docstring in files within the ``docs/`` directory. - -.. _autodoc: http://www.sphinx-doc.org/en/stable/ext/autodoc.html +In any case, if something is covered in a docstring, use the +:mod:`~sphinx.ext.autodoc` extension to pull the docstring into the +documentation instead of duplicating the docstring in files within the +``docs/`` directory. Tests ===== From 8a6a063778d45e8ba68a59558d2930332c1e9a83 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 12 Nov 2019 10:23:19 +0100 Subject: [PATCH 088/113] Allow opening the source code from the API documentation --- docs/conf.py | 1 + 1 file changed, 1 insertion(+) diff --git a/docs/conf.py b/docs/conf.py index 34dd5bcb7..b09000de0 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -31,6 +31,7 @@ extensions = [ 'sphinx.ext.autodoc', 'sphinx.ext.coverage', 'sphinx.ext.intersphinx', + 'sphinx.ext.viewcode', ] # Add any paths that contain templates here, relative to this directory. From 4b8b0345e58ee5990bd0c28205b5eb0b892680d1 Mon Sep 17 00:00:00 2001 From: purvaudai Date: Tue, 12 Nov 2019 18:17:15 +0530 Subject: [PATCH 089/113] Mades Changes as per review --- requirements-py3.txt | 1 - tests/test_feedexport.py | 2 +- 2 files changed, 1 insertion(+), 2 deletions(-) diff --git a/requirements-py3.txt b/requirements-py3.txt index 77296b91b..2c98e6f6d 100644 --- a/requirements-py3.txt +++ b/requirements-py3.txt @@ -16,4 +16,3 @@ lxml>=3.5.0 service_identity>=16.0.0 six>=1.10.0 zope.interface>=4.1.3 -pathlib2>=2.0 diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 2819f8f0b..1f46ac04a 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -29,7 +29,7 @@ from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get from scrapy.utils.python import to_native_str from scrapy.utils.project import get_project_settings -from pathlib2 import Path +from pathlib import Path class FileFeedStorageTest(unittest.TestCase): From 414e6e2fd568e0dfae699873d3c1ccd865261d09 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 13 Nov 2019 07:56:45 +0100 Subject: [PATCH 090/113] Skip a doctest in Python 3.5- because of dictionary changes --- docs/intro/tutorial.rst | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 996e3b475..30b1ddeab 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -402,6 +402,12 @@ to get all of them:: >>> tags ['change', 'deep-thoughts', 'thinking', 'world'] +.. invisible-code-block: python + + from sys import version_info + +.. skip: next if(version_info <= (3, 5), reason="Only Python 3.6+ dictionaries match the output") + Having figured out how to extract each bit, we can now iterate over all the quotes elements and put them together into a Python dictionary:: From b642a1fca29852adf0ba3ddd62c2ecfdbaf9610e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 13 Nov 2019 09:14:20 +0100 Subject: [PATCH 091/113] Fix doctest skipping based on the running Python version --- docs/intro/tutorial.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 30b1ddeab..6b15a5fbd 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -406,7 +406,7 @@ to get all of them:: from sys import version_info -.. skip: next if(version_info <= (3, 5), reason="Only Python 3.6+ dictionaries match the output") +.. skip: next if(version_info < (3, 6), reason="Only Python 3.6+ dictionaries match the output") Having figured out how to extract each bit, we can now iterate over all the quotes elements and put them together into a Python dictionary:: From 76c31094dff2920778b42ed746c6a9cd5b08f4e4 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 13 Nov 2019 09:28:48 +0100 Subject: [PATCH 092/113] Install the sphinx-notfound-page Sphinx extension --- docs/conf.py | 1 + docs/requirements.txt | 3 ++- 2 files changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/conf.py b/docs/conf.py index 6ab5959d5..935c3c9a1 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -27,6 +27,7 @@ sys.path.insert(0, path.dirname(path.dirname(__file__))) # Add any Sphinx extension module names here, as strings. They can be extensions # coming with Sphinx (named 'sphinx.ext.*') or your custom ones. extensions = [ + 'notfound.extension', 'scrapydocs', 'sphinx.ext.autodoc', 'sphinx.ext.coverage', diff --git a/docs/requirements.txt b/docs/requirements.txt index 379da9994..f9db85146 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1,2 +1,3 @@ Sphinx>=2.1 -sphinx_rtd_theme \ No newline at end of file +sphinx-notfound-page +sphinx_rtd_theme From 33ef24c757c797c804c8fd242b8ab5705219b452 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 13 Nov 2019 10:52:05 +0100 Subject: [PATCH 093/113] =?UTF-8?q?Add=20missing=20whitespace=20after=20?= =?UTF-8?q?=E2=80=98,=E2=80=99,=20=E2=80=98;=E2=80=99=20or=20=E2=80=98:?= =?UTF-8?q?=E2=80=99?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- pytest.ini | 16 ++++++++-------- scrapy/core/spidermw.py | 2 +- tests/test_http_request.py | 4 ++-- tests/test_item.py | 2 +- tests/test_linkextractors.py | 4 ++-- tests/test_logformatter.py | 2 +- tests/test_utils_conf.py | 2 +- tests/test_utils_console.py | 2 +- tests/test_utils_misc/__init__.py | 2 +- 9 files changed, 18 insertions(+), 18 deletions(-) diff --git a/pytest.ini b/pytest.ini index 8c5a2cd54..984930459 100644 --- a/pytest.ini +++ b/pytest.ini @@ -48,7 +48,7 @@ flake8-ignore = scrapy/core/engine.py E261 E501 E128 E127 E306 E502 scrapy/core/scheduler.py E501 scrapy/core/scraper.py E501 E306 E261 E128 W504 - scrapy/core/spidermw.py E501 E731 E502 E231 E126 E226 + scrapy/core/spidermw.py E501 E731 E502 E126 E226 scrapy/core/downloader/__init__.py F401 E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 @@ -214,13 +214,13 @@ flake8-ignore = tests/test_feedexport.py E501 F401 F841 E241 tests/test_http_cookies.py E501 tests/test_http_headers.py E302 E501 - tests/test_http_request.py F401 E402 E501 E231 E261 E127 E128 W293 E502 E128 E502 E126 E123 + tests/test_http_request.py F401 E402 E501 E261 E127 E128 W293 E502 E128 E502 E126 E123 tests/test_http_response.py E501 E301 E502 E128 E265 - tests/test_item.py E701 E128 E231 F841 E306 + tests/test_item.py E701 E128 F841 E306 tests/test_link.py E501 - tests/test_linkextractors.py E501 E128 E231 E124 + tests/test_linkextractors.py E501 E128 E124 tests/test_loader.py E302 E501 E731 E303 E741 E128 E117 E241 - tests/test_logformatter.py E128 E501 E231 E122 E302 + tests/test_logformatter.py E128 E501 E122 E302 tests/test_mail.py E302 E128 E501 E305 tests/test_middleware.py E302 E501 E128 tests/test_pipeline_crawl.py E131 E501 E128 E126 @@ -239,8 +239,8 @@ flake8-ignore = tests/test_spidermiddleware_output_chain.py F401 E501 E302 W293 E226 tests/test_spidermiddleware_referer.py F401 E501 E302 F841 E125 E201 E261 E124 E501 E241 E121 tests/test_squeues.py E501 E302 E701 E741 - tests/test_utils_conf.py E501 E231 E303 E128 - tests/test_utils_console.py E302 E231 + tests/test_utils_conf.py E501 E303 E128 + tests/test_utils_console.py E302 tests/test_utils_curl.py E501 tests/test_utils_datatypes.py E402 E501 E305 tests/test_utils_defer.py E306 E261 E501 E302 F841 E226 @@ -269,4 +269,4 @@ flake8-ignore = tests/test_spiderloader/test_spiders/spider2.py E302 tests/test_spiderloader/test_spiders/spider3.py E302 tests/test_spiderloader/test_spiders/nested/spider4.py E302 - tests/test_utils_misc/__init__.py E501 E231 + tests/test_utils_misc/__init__.py E501 diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index b5f9837ff..00cee3ada 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -36,7 +36,7 @@ class SpiderMiddlewareManager(MiddlewareManager): self.methods['process_spider_exception'].appendleft(getattr(mw, 'process_spider_exception', None)) def scrape_response(self, scrape_func, response, request, spider): - fname = lambda f:'%s.%s' % ( + fname = lambda f: '%s.%s' % ( six.get_method_self(f).__class__.__name__, six.get_method_function(f).__name__) diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 96a4fb141..45a547f40 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -56,7 +56,7 @@ class RequestTest(unittest.TestCase): def test_headers(self): # Different ways of setting headers attribute url = 'http://www.scrapy.org' - headers = {b'Accept':'gzip', b'Custom-Header':'nothing to tell you'} + headers = {b'Accept': 'gzip', b'Custom-Header': 'nothing to tell you'} r = self.request_class(url=url, headers=headers) p = self.request_class(url=url, headers=r.headers) @@ -816,7 +816,7 @@ class FormRequestTest(RequestTest): """) - r1 = self.request_class.from_response(response, formdata={'two':'3'}) + r1 = self.request_class.from_response(response, formdata={'two': '3'}) self.assertEqual(r1.method, 'POST') self.assertEqual(r1.headers['Content-type'], b'application/x-www-form-urlencoded') fs = _qs(r1) diff --git a/tests/test_item.py b/tests/test_item.py index 947566686..7c9468f65 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -245,7 +245,7 @@ class ItemTest(unittest.TestCase): def test_copy(self): class TestItem(Item): name = Field() - item = TestItem({'name':'lower'}) + item = TestItem({'name': 'lower'}) copied_item = item.copy() self.assertNotEqual(id(item), id(copied_item)) copied_item['name'] = copied_item['name'].upper() diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index d96e259f6..57ef1694a 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -322,7 +322,7 @@ class Base: Link(url=page4_url, text=u'href with whitespaces'), ]) - lx = self.extractor_cls(attrs=("href","src"), tags=("a","area","img"), deny_extensions=()) + lx = self.extractor_cls(attrs=("href", "src"), tags=("a", "area", "img"), deny_extensions=()) self.assertEqual(lx.extract_links(self.response), [ Link(url='http://example.com/sample1.html', text=u''), Link(url='http://example.com/sample2.html', text=u'sample 2'), @@ -360,7 +360,7 @@ class Base: Link(url='http://example.com/sample2.html', text=u'sample 2'), ]) - lx = self.extractor_cls(tags=("a","img"), attrs=("href", "src"), deny_extensions=()) + lx = self.extractor_cls(tags=("a", "img"), attrs=("href", "src"), deny_extensions=()) self.assertEqual(lx.extract_links(response), [ Link(url='http://example.com/sample2.html', text=u'sample 2'), Link(url='http://example.com/sample2.jpg', text=u''), diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index eb9c4a561..b4ea30bb7 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -45,7 +45,7 @@ class LoggingContribTest(unittest.TestCase): "Crawled (200) (referer: http://example.com) ['cached']") def test_flags_in_request(self): - req = Request("http://www.example.com", flags=['test','flag']) + req = Request("http://www.example.com", flags=['test', 'flag']) res = Response("http://www.example.com") logkws = self.formatter.crawled(req, res, self.spider) logline = logkws['msg'] % logkws['args'] diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index 29937c189..02d8ba51e 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -79,7 +79,7 @@ class BuildComponentListTest(unittest.TestCase): self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) d = {'one': {'a': 'a', 'b': 2}} self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) - d = {'one': 'lorem ipsum',} + d = {'one': 'lorem ipsum'} self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) diff --git a/tests/test_utils_console.py b/tests/test_utils_console.py index 65782747b..c2211848c 100644 --- a/tests/test_utils_console.py +++ b/tests/test_utils_console.py @@ -21,7 +21,7 @@ class UtilsConsoleTestCase(unittest.TestCase): shell = get_shell_embed_func(['invalid']) self.assertEqual(shell, None) - shell = get_shell_embed_func(['invalid','python']) + shell = get_shell_embed_func(['invalid', 'python']) self.assertTrue(callable(shell)) self.assertEqual(shell.__name__, '_embed_standard_shell') diff --git a/tests/test_utils_misc/__init__.py b/tests/test_utils_misc/__init__.py index e109d5343..de6f173e0 100644 --- a/tests/test_utils_misc/__init__.py +++ b/tests/test_utils_misc/__init__.py @@ -74,7 +74,7 @@ class UtilsMiscTestCase(unittest.TestCase): self.assertEqual(list(arg_to_iter(100)), [100]) self.assertEqual(list(arg_to_iter(l for l in 'abc')), ['a', 'b', 'c']) self.assertEqual(list(arg_to_iter([1, 2, 3])), [1, 2, 3]) - self.assertEqual(list(arg_to_iter({'a':1})), [{'a': 1}]) + self.assertEqual(list(arg_to_iter({'a': 1})), [{'a': 1}]) self.assertEqual(list(arg_to_iter(TestItem(name="john"))), [TestItem(name="john")]) def test_create_instance(self): From 1d7c8cb0b1d3aa225fd396f263938fd2f171fb73 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Mon, 22 Jul 2019 22:27:29 +0500 Subject: [PATCH 094/113] Remove six.PY2 and six.PY3 conditionals. --- docs/topics/downloader-middleware.rst | 6 +++--- scrapy/_monkeypatches.py | 10 ---------- scrapy/commands/fetch.py | 5 ++--- scrapy/crawler.py | 11 ----------- scrapy/exporters.py | 2 +- scrapy/extensions/feedexport.py | 3 +-- scrapy/http/request/form.py | 3 +-- scrapy/http/response/text.py | 3 --- scrapy/item.py | 8 +------- scrapy/link.py | 15 ++------------- scrapy/mail.py | 9 ++------- scrapy/settings/__init__.py | 8 +------- scrapy/settings/default_settings.py | 4 +--- scrapy/utils/boto.py | 10 +--------- scrapy/utils/conf.py | 5 +---- scrapy/utils/datatypes.py | 20 +++++++------------- scrapy/utils/gz.py | 13 +++---------- scrapy/utils/iterators.py | 6 +----- scrapy/utils/python.py | 10 ++-------- tests/__init__.py | 15 --------------- tests/test_http_request.py | 11 +++-------- tests/test_http_response.py | 3 +-- tests/test_item.py | 8 ++------ tests/test_link.py | 13 ++----------- tests/test_middleware.py | 21 ++++++--------------- tests/test_request_cb_kwargs.py | 8 +------- tests/test_settings/__init__.py | 8 +------- tests/test_utils_datatypes.py | 7 +------ tests/test_utils_python.py | 7 +++---- 29 files changed, 50 insertions(+), 202 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 8048e1c86..366b95510 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -474,7 +474,7 @@ DBM storage backend A DBM_ storage backend is also available for the HTTP cache middleware. - By default, it uses the anydbm_ module, but you can change it with the + By default, it uses the dbm_ module, but you can change it with the :setting:`HTTPCACHE_DBM_MODULE` setting. .. _httpcache-storage-custom: @@ -626,7 +626,7 @@ HTTPCACHE_DBM_MODULE .. versionadded:: 0.13 -Default: ``'anydbm'`` +Default: ``'dbm'`` The database module to use in the :ref:`DBM storage backend `. This setting is specific to the DBM backend. @@ -1202,4 +1202,4 @@ The default encoding for proxy authentication on :class:`HttpProxyMiddleware`. .. _DBM: https://en.wikipedia.org/wiki/Dbm -.. _anydbm: https://docs.python.org/2/library/anydbm.html +.. _dbm: https://docs.python.org/3/library/dbm.html diff --git a/scrapy/_monkeypatches.py b/scrapy/_monkeypatches.py index b68099cad..1f8067b35 100644 --- a/scrapy/_monkeypatches.py +++ b/scrapy/_monkeypatches.py @@ -1,16 +1,6 @@ -import six from six.moves import copyreg -if six.PY2: - from urlparse import urlparse - - # workaround for https://bugs.python.org/issue9374 - Python < 2.7.4 - if urlparse('s3://bucket/key?key=value').query != 'key=value': - from urlparse import uses_query - uses_query.append('s3') - - # Undo what Twisted's perspective broker adds to pickle register # to prevent bugs like Twisted#7989 while serializing requests import twisted.persisted.styles # NOQA diff --git a/scrapy/commands/fetch.py b/scrapy/commands/fetch.py index 7d4840529..d45133e0e 100644 --- a/scrapy/commands/fetch.py +++ b/scrapy/commands/fetch.py @@ -1,5 +1,5 @@ from __future__ import print_function -import sys, six +import sys from w3lib.url import is_url from scrapy.commands import ScrapyCommand @@ -45,8 +45,7 @@ class Command(ScrapyCommand): self._print_bytes(response.body) def _print_bytes(self, bytes_): - bytes_writer = sys.stdout if six.PY2 else sys.stdout.buffer - bytes_writer.write(bytes_ + b'\n') + sys.stdout.buffer.write(bytes_ + b'\n') def run(self, args, opts): if len(args) != 1 or not is_url(args[0]): diff --git a/scrapy/crawler.py b/scrapy/crawler.py index ded3c082b..19b998e0d 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -88,20 +88,9 @@ class Crawler(object): yield self.engine.open_spider(self.spider, start_requests) yield defer.maybeDeferred(self.engine.start) except Exception: - # In Python 2 reraising an exception after yield discards - # the original traceback (see https://bugs.python.org/issue7563), - # so sys.exc_info() workaround is used. - # This workaround also works in Python 3, but it is not needed, - # and it is slower, so in Python 3 we use native `raise`. - if six.PY2: - exc_info = sys.exc_info() - self.crawling = False if self.engine is not None: yield self.engine.close() - - if six.PY2: - six.reraise(*exc_info) raise def _create_spider(self, *args, **kwargs): diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 8ed8d55f1..0d9c35654 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -216,7 +216,7 @@ class CsvItemExporter(BaseItemExporter): write_through=True, encoding=self.encoding, newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034 - ) if six.PY3 else file + ) self.csv_writer = csv.writer(self.stream, **kwargs) self._headers_not_written = True self._join_multivalued = join_multivalued diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 41d68fb14..e2492d506 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -10,7 +10,6 @@ import logging import posixpath from tempfile import NamedTemporaryFile from datetime import datetime -import six from six.moves.urllib.parse import urlparse, unquote from ftplib import FTP @@ -65,7 +64,7 @@ class StdoutFeedStorage(object): def __init__(self, uri, _stdout=None): if not _stdout: - _stdout = sys.stdout if six.PY2 else sys.stdout.buffer + _stdout = sys.stdout.buffer self._stdout = _stdout def open(self, spider): diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py index 3ce8fc48e..b6feede07 100644 --- a/scrapy/http/request/form.py +++ b/scrapy/http/request/form.py @@ -104,8 +104,7 @@ def _get_form(response, formname, formid, formnumber, formxpath): el = el.getparent() if el is None: break - encoded = formxpath if six.PY3 else formxpath.encode('unicode_escape') - raise ValueError('No
element found with %s' % encoded) + raise ValueError('No element found with %s' % formxpath) # If we get here, it means that either formname was None # or invalid diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 339913d4e..a8010877c 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -32,9 +32,6 @@ class TextResponse(Response): def _set_url(self, url): if isinstance(url, six.text_type): - if six.PY2 and self.encoding is None: - raise TypeError("Cannot convert unicode url - %s " - "has no encoding" % type(self).__name__) self._url = to_native_str(url, self.encoding) else: super(TextResponse, self)._set_url(url) diff --git a/scrapy/item.py b/scrapy/item.py index 73b8f54b0..32f9b2ebb 100644 --- a/scrapy/item.py +++ b/scrapy/item.py @@ -4,8 +4,8 @@ Scrapy Item See documentation in docs/topics/item.rst """ -import collections from abc import ABCMeta +from collections.abc import MutableMapping from copy import deepcopy from pprint import pformat from warnings import warn @@ -16,12 +16,6 @@ from scrapy.utils.deprecate import ScrapyDeprecationWarning from scrapy.utils.trackref import object_ref -if six.PY2: - MutableMapping = collections.MutableMapping -else: - MutableMapping = collections.abc.MutableMapping - - class BaseItem(object_ref): """Base class for all scraped items. diff --git a/scrapy/link.py b/scrapy/link.py index f0638ced2..be1888ef0 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -4,12 +4,6 @@ This module defines the Link object used in Link extractors. For actual link extractors implementation see scrapy.linkextractors, or its documentation in: docs/topics/link-extractors.rst """ -import warnings -import six - -from scrapy.utils.python import to_bytes - - class Link(object): """Link objects represent an extracted link by the LinkExtractor.""" @@ -17,13 +11,8 @@ class Link(object): def __init__(self, url, text='', fragment='', nofollow=False): if not isinstance(url, str): - if six.PY2: - warnings.warn("Link urls must be str objects. " - "Assuming utf-8 encoding (which could be wrong)") - url = to_bytes(url, encoding='utf8') - else: - got = url.__class__.__name__ - raise TypeError("Link urls must be str objects, got %s" % got) + got = url.__class__.__name__ + raise TypeError("Link urls must be str objects, got %s" % got) self.url = url self.text = text self.fragment = fragment diff --git a/scrapy/mail.py b/scrapy/mail.py index 5b944e1c4..746468e25 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -9,18 +9,13 @@ try: from cStringIO import StringIO as BytesIO except ImportError: from io import BytesIO -import six from email.utils import COMMASPACE, formatdate from six.moves.email_mime_multipart import MIMEMultipart from six.moves.email_mime_text import MIMEText from six.moves.email_mime_base import MIMEBase -if six.PY2: - from email.MIMENonMultipart import MIMENonMultipart - from email import Encoders -else: - from email.mime.nonmultipart import MIMENonMultipart - from email import encoders as Encoders +from email.mime.nonmultipart import MIMENonMultipart +from email import encoders as Encoders from twisted.internet import defer, reactor, ssl diff --git a/scrapy/settings/__init__.py b/scrapy/settings/__init__.py index f28c7940d..c871e86e0 100644 --- a/scrapy/settings/__init__.py +++ b/scrapy/settings/__init__.py @@ -1,19 +1,13 @@ import six import json import copy -import collections +from collections.abc import MutableMapping from importlib import import_module from pprint import pformat from scrapy.settings import default_settings -if six.PY2: - MutableMapping = collections.MutableMapping -else: - MutableMapping = collections.abc.MutableMapping - - SETTINGS_PRIORITIES = { 'default': 0, 'command': 10, diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index 9c22999cb..5c9678c01 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -17,8 +17,6 @@ import sys from importlib import import_module from os.path import join, abspath, dirname -import six - AJAXCRAWL_ENABLED = False AUTOTHROTTLE_ENABLED = False @@ -179,7 +177,7 @@ HTTPCACHE_ALWAYS_STORE = False HTTPCACHE_IGNORE_HTTP_CODES = [] HTTPCACHE_IGNORE_SCHEMES = ['file'] HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS = [] -HTTPCACHE_DBM_MODULE = 'anydbm' if six.PY2 else 'dbm' +HTTPCACHE_DBM_MODULE = 'dbm' HTTPCACHE_POLICY = 'scrapy.extensions.httpcache.DummyPolicy' HTTPCACHE_GZIP = False diff --git a/scrapy/utils/boto.py b/scrapy/utils/boto.py index 421ab2f7e..c8fc911bb 100644 --- a/scrapy/utils/boto.py +++ b/scrapy/utils/boto.py @@ -1,7 +1,6 @@ """Boto/botocore helpers""" from __future__ import absolute_import -import six from scrapy.exceptions import NotConfigured @@ -11,11 +10,4 @@ def is_botocore(): import botocore return True except ImportError: - if six.PY2: - try: - import boto - return False - except ImportError: - raise NotConfigured('missing botocore or boto library') - else: - raise NotConfigured('missing botocore library') + raise NotConfigured('missing botocore library') diff --git a/scrapy/utils/conf.py b/scrapy/utils/conf.py index fb7ca3310..561bb72fc 100644 --- a/scrapy/utils/conf.py +++ b/scrapy/utils/conf.py @@ -1,13 +1,10 @@ +from configparser import ConfigParser import os import sys import numbers from operator import itemgetter import six -if six.PY2: - from ConfigParser import SafeConfigParser as ConfigParser -else: - from configparser import ConfigParser from scrapy.settings import BaseSettings from scrapy.utils.deprecate import update_classpath diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index 87536e9d7..39d389fa6 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -7,6 +7,7 @@ This module must not depend on any module outside the Standard Library. import copy import collections +from collections.abc import Mapping import warnings import six @@ -14,12 +15,6 @@ import six from scrapy.exceptions import ScrapyDeprecationWarning -if six.PY2: - Mapping = collections.Mapping -else: - Mapping = collections.abc.Mapping - - class MultiValueDictKeyError(KeyError): def __init__(self, *args, **kwargs): warnings.warn( @@ -252,13 +247,12 @@ class MergeDict(object): first occurrence will be used. """ def __init__(self, *dicts): - if not six.PY2: - warnings.warn( - "scrapy.utils.datatypes.MergeDict is deprecated in favor " - "of collections.ChainMap (introduced in Python 3.3)", - category=ScrapyDeprecationWarning, - stacklevel=2, - ) + warnings.warn( + "scrapy.utils.datatypes.MergeDict is deprecated in favor " + "of collections.ChainMap (introduced in Python 3.3)", + category=ScrapyDeprecationWarning, + stacklevel=2, + ) self.dicts = dicts def __getitem__(self, key): diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index b3fb16b1e..9984492f0 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -6,7 +6,6 @@ except ImportError: from io import BytesIO from gzip import GzipFile -import six import re from scrapy.utils.decorators import deprecated @@ -17,14 +16,8 @@ from scrapy.utils.decorators import deprecated # (regression or bug-fix compared to Python 3.4) # - read1(), which fetches data before raising EOFError on next call # works here but is only available from Python>=3.3 -# - scrapy does not support Python 3.2 -# - Python 2.7 GzipFile works fine with standard read() + extrabuf -if six.PY2: - def read1(gzf, size=-1): - return gzf.read(size) -else: - def read1(gzf, size=-1): - return gzf.read1(size) +def read1(gzf, size=-1): + return gzf.read1(size) def gunzip(data): @@ -37,7 +30,7 @@ def gunzip(data): chunk = b'.' while chunk: try: - chunk = read1(f, 8196) + chunk = f.read1(8196) output_list.append(chunk) except (IOError, EOFError, struct.error): # complete only if there is some data, otherwise re-raise diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index a12e14005..dbc1e0d20 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -102,11 +102,7 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): def row_to_unicode(row_): return [to_unicode(field, encoding) for field in row_] - # Python 3 csv reader input object needs to return strings - if six.PY3: - lines = StringIO(_body_or_str(obj, unicode=True)) - else: - lines = BytesIO(_body_or_str(obj, unicode=False)) + lines = StringIO(_body_or_str(obj, unicode=True)) kwargs = {} if delimiter: kwargs["delimiter"] = delimiter diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index ea5193f12..37e6be868 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -113,10 +113,7 @@ def to_bytes(text, encoding=None, errors='strict'): def to_native_str(text, encoding=None, errors='strict'): """ Return str representation of ``text`` (bytes in Python 2.x and unicode in Python 3.x). """ - if six.PY2: - return to_bytes(text, encoding, errors) - else: - return to_unicode(text, encoding, errors) + return to_unicode(text, encoding, errors) def re_rsearch(pattern, text, chunk_size=1024): @@ -189,7 +186,7 @@ def _getargspec_py23(func): """_getargspec_py23(function) -> named tuple ArgSpec(args, varargs, keywords, defaults) - Identical to inspect.getargspec() in python2, but uses + Was identical to inspect.getargspec() in python2, but uses inspect.getfullargspec() for python3 behind the scenes to avoid DeprecationWarning. @@ -199,9 +196,6 @@ def _getargspec_py23(func): >>> _getargspec_py23(f) ArgSpec(args=['a', 'b'], varargs='ar', keywords='kw', defaults=(2,)) """ - if six.PY2: - return inspect.getargspec(func) - return inspect.ArgSpec(*inspect.getfullargspec(func)[:4]) diff --git a/tests/__init__.py b/tests/__init__.py index 9c9e35c35..a54367f8c 100644 --- a/tests/__init__.py +++ b/tests/__init__.py @@ -35,18 +35,3 @@ def get_testdata(*paths): path = os.path.join(tests_datadir, *paths) with open(path, 'rb') as f: return f.read() - - -# FIXME: delete after dropping py2 support -# Monkey patch the unittest module to prevent the -# DeprecationWarning about assertRaisesRegexp -> assertRaisesRegex -import six -if six.PY2: - import unittest - import twisted.trial.unittest - if not getattr(unittest.TestCase, 'assertRegex', None): - unittest.TestCase.assertRegex = unittest.TestCase.assertRegexpMatches - if not getattr(unittest.TestCase, 'assertRaisesRegex', None): - unittest.TestCase.assertRaisesRegex = unittest.TestCase.assertRaisesRegexp - if not getattr(twisted.trial.unittest.TestCase, 'assertRaisesRegex', None): - twisted.trial.unittest.TestCase.assertRaisesRegex = twisted.trial.unittest.TestCase.assertRaisesRegexp diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 96a4fb141..bb451b5f4 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -3,13 +3,12 @@ import cgi import unittest import re import json +from urllib.parse import unquote_to_bytes import warnings import six from six.moves import xmlrpc_client as xmlrpclib from six.moves.urllib.parse import urlparse, parse_qs, unquote -if six.PY3: - from urllib.parse import unquote_to_bytes from scrapy.http import Request, FormRequest, XmlRpcRequest, JsonRequest, Headers, HtmlResponse from scrapy.utils.python import to_bytes, to_native_str @@ -1064,8 +1063,7 @@ class FormRequestTest(RequestTest): self.assertEqual(fs, {}) xpath = u"//form[@name='\u03b1']" - encoded = xpath if six.PY3 else xpath.encode('unicode_escape') - self.assertRaisesRegex(ValueError, re.escape(encoded), + self.assertRaisesRegex(ValueError, re.escape(xpath), self.request_class.from_response, response, formxpath=xpath) @@ -1208,10 +1206,7 @@ def _qs(req, encoding='utf-8', to_unicode=False): qs = req.body else: qs = req.url.partition('?')[2] - if six.PY2: - uqs = unquote(to_native_str(qs, encoding)) - elif six.PY3: - uqs = unquote_to_bytes(qs) + uqs = unquote_to_bytes(qs) if to_unicode: uqs = uqs.decode(encoding) return parse_qs(uqs, True) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index dfc8562f3..ec3b51086 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -21,8 +21,7 @@ class BaseResponseTest(unittest.TestCase): # Response requires url in the consturctor self.assertRaises(Exception, self.response_class) self.assertTrue(isinstance(self.response_class('http://example.com/'), self.response_class)) - if not six.PY2: - self.assertRaises(TypeError, self.response_class, b"http://example.com") + self.assertRaises(TypeError, self.response_class, b"http://example.com") # body can be str or None self.assertTrue(isinstance(self.response_class('http://example.com/', body=b''), self.response_class)) self.assertTrue(isinstance(self.response_class('http://example.com/', body=b'body'), self.response_class)) diff --git a/tests/test_item.py b/tests/test_item.py index 947566686..0ad278701 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -62,12 +62,8 @@ class ItemTest(unittest.TestCase): i['number'] = 123 itemrepr = repr(i) - if six.PY2: - self.assertEqual(itemrepr, - "{'name': u'John Doe', 'number': 123}") - else: - self.assertEqual(itemrepr, - "{'name': 'John Doe', 'number': 123}") + self.assertEqual(itemrepr, + "{'name': 'John Doe', 'number': 123}") i2 = eval(itemrepr) self.assertEqual(i2['name'], 'John Doe') diff --git a/tests/test_link.py b/tests/test_link.py index 955430b37..5e2ce5eeb 100644 --- a/tests/test_link.py +++ b/tests/test_link.py @@ -1,6 +1,4 @@ import unittest -import warnings -import six from scrapy.link import Link @@ -46,12 +44,5 @@ class LinkTest(unittest.TestCase): self._assert_same_links(l1, l2) def test_non_str_url_py2(self): - if six.PY2: - with warnings.catch_warnings(record=True) as w: - link = Link(u"http://www.example.com/\xa3") - self.assertIsInstance(link.url, str) - self.assertEqual(link.url, b'http://www.example.com/\xc2\xa3') - assert len(w) == 1, "warning not issued" - else: - with self.assertRaises(TypeError): - Link(b"http://www.example.com/\xc2\xa3") + with self.assertRaises(TypeError): + Link(b"http://www.example.com/\xc2\xa3") diff --git a/tests/test_middleware.py b/tests/test_middleware.py index aea0be825..af9b43d61 100644 --- a/tests/test_middleware.py +++ b/tests/test_middleware.py @@ -3,7 +3,6 @@ from twisted.trial import unittest from scrapy.settings import Settings from scrapy.exceptions import NotConfigured from scrapy.middleware import MiddlewareManager -import six class M1(object): @@ -66,20 +65,12 @@ class MiddlewareManagerTest(unittest.TestCase): def test_methods(self): mwman = TestMiddlewareManager(M1(), M2(), M3()) - if six.PY2: - self.assertEqual([x.im_class for x in mwman.methods['open_spider']], - [M1, M2]) - self.assertEqual([x.im_class for x in mwman.methods['close_spider']], - [M2, M1]) - self.assertEqual([x.im_class for x in mwman.methods['process']], - [M1, M3]) - else: - self.assertEqual([x.__self__.__class__ for x in mwman.methods['open_spider']], - [M1, M2]) - self.assertEqual([x.__self__.__class__ for x in mwman.methods['close_spider']], - [M2, M1]) - self.assertEqual([x.__self__.__class__ for x in mwman.methods['process']], - [M1, M3]) + self.assertEqual([x.__self__.__class__ for x in mwman.methods['open_spider']], + [M1, M2]) + self.assertEqual([x.__self__.__class__ for x in mwman.methods['close_spider']], + [M2, M1]) + self.assertEqual([x.__self__.__class__ for x in mwman.methods['process']], + [M1, M3]) def test_enabled(self): m1, m2, m3 = M1(), M2(), M3() diff --git a/tests/test_request_cb_kwargs.py b/tests/test_request_cb_kwargs.py index c9943faa8..a5cdc0de0 100644 --- a/tests/test_request_cb_kwargs.py +++ b/tests/test_request_cb_kwargs.py @@ -1,7 +1,6 @@ from testfixtures import LogCapture from twisted.internet import defer from twisted.trial.unittest import TestCase -import six from scrapy.http import Request from scrapy.crawler import CrawlerRunner @@ -161,9 +160,4 @@ class CallbackKeywordArgumentsTestCase(TestCase): self.assertEqual(exceptions['takes_less'].exc_info[0], TypeError) self.assertEqual(str(exceptions['takes_less'].exc_info[1]), "parse_takes_less() got an unexpected keyword argument 'number'") self.assertEqual(exceptions['takes_more'].exc_info[0], TypeError) - # py2 and py3 messages are different - exc_message = str(exceptions['takes_more'].exc_info[1]) - if six.PY2: - self.assertEqual(exc_message, "parse_takes_more() takes exactly 5 arguments (4 given)") - elif six.PY3: - self.assertEqual(exc_message, "parse_takes_more() missing 1 required positional argument: 'other'") + self.assertEqual(str(exceptions['takes_more'].exc_info[1]), "parse_takes_more() missing 1 required positional argument: 'other'") diff --git a/tests/test_settings/__init__.py b/tests/test_settings/__init__.py index 1dbacbea3..08286ff02 100644 --- a/tests/test_settings/__init__.py +++ b/tests/test_settings/__init__.py @@ -60,9 +60,6 @@ class SettingsAttributeTest(unittest.TestCase): class BaseSettingsTest(unittest.TestCase): - if six.PY3: - assertItemsEqual = unittest.TestCase.assertCountEqual - def setUp(self): self.settings = BaseSettings() @@ -152,7 +149,7 @@ class BaseSettingsTest(unittest.TestCase): self.settings.setmodule( 'tests.test_settings.default_settings', 10) - self.assertItemsEqual(six.iterkeys(self.settings.attributes), + self.assertCountEqual(six.iterkeys(self.settings.attributes), six.iterkeys(ctrl_attributes)) for key in six.iterkeys(ctrl_attributes): @@ -343,9 +340,6 @@ class BaseSettingsTest(unittest.TestCase): class SettingsTest(unittest.TestCase): - if six.PY3: - assertItemsEqual = unittest.TestCase.assertCountEqual - def setUp(self): self.settings = Settings() diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 7e671f627..53228fc6e 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -1,11 +1,6 @@ import copy import unittest - -import six -if six.PY2: - from collections import Mapping, MutableMapping -else: - from collections.abc import Mapping, MutableMapping +from collections.abc import Mapping, MutableMapping from scrapy.utils.datatypes import CaselessDict, LocalCache, SequenceExclude diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index 3e1148354..096aa50b7 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -231,12 +231,11 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertEqual(get_func_args(" ".join), []) self.assertEqual(get_func_args(operator.itemgetter(2)), []) else: - stripself = not six.PY2 # PyPy3 exposes them as methods self.assertEqual( - get_func_args(six.text_type.split, stripself), ['sep', 'maxsplit']) - self.assertEqual(get_func_args(" ".join, stripself), ['list']) + get_func_args(six.text_type.split, True), ['sep', 'maxsplit']) + self.assertEqual(get_func_args(" ".join, True), ['list']) self.assertEqual( - get_func_args(operator.itemgetter(2), stripself), ['obj']) + get_func_args(operator.itemgetter(2), True), ['obj']) def test_without_none_values(self): From 0e696ed06d2975ee86e7ce3a2d3892588c420a04 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 20 Aug 2019 20:56:26 +0500 Subject: [PATCH 095/113] Remove unneeded and unused code from XmlItemExporter. --- scrapy/exporters.py | 22 ++++------------------ 1 file changed, 4 insertions(+), 18 deletions(-) diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 0d9c35654..1aa195b0b 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -143,11 +143,11 @@ class XmlItemExporter(BaseItemExporter): def _beautify_newline(self, new_item=False): if self.indent is not None and (self.indent > 0 or new_item): - self._xg_characters('\n') + self.xg.characters('\n') def _beautify_indent(self, depth=1): if self.indent: - self._xg_characters(' ' * self.indent * depth) + self.xg.characters(' ' * self.indent * depth) def start_exporting(self): self.xg.startDocument() @@ -182,26 +182,12 @@ class XmlItemExporter(BaseItemExporter): self._export_xml_field('value', value, depth=depth+1) self._beautify_indent(depth=depth) elif isinstance(serialized_value, six.text_type): - self._xg_characters(serialized_value) + self.xg.characters(serialized_value) else: - self._xg_characters(str(serialized_value)) + self.xg.characters(str(serialized_value)) self.xg.endElement(name) self._beautify_newline() - # Workaround for https://bugs.python.org/issue17606 - # Before Python 2.7.4 xml.sax.saxutils required bytes; - # since 2.7.4 it requires unicode. The bug is likely to be - # fixed in 2.7.6, but 2.7.6 will still support unicode, - # and Python 3.x will require unicode, so ">= 2.7.4" should be fine. - if sys.version_info[:3] >= (2, 7, 4): - def _xg_characters(self, serialized_value): - if not isinstance(serialized_value, six.text_type): - serialized_value = serialized_value.decode(self.encoding) - return self.xg.characters(serialized_value) - else: # pragma: no cover - def _xg_characters(self, serialized_value): - return self.xg.characters(serialized_value) - class CsvItemExporter(BaseItemExporter): From 065fe29d3cc6634893dfead320779498fb061cd4 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 20 Aug 2019 21:06:52 +0500 Subject: [PATCH 096/113] Deprecate scrapy.utils.gz.read1. --- scrapy/utils/gz.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index 9984492f0..dc8316d8c 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -16,6 +16,7 @@ from scrapy.utils.decorators import deprecated # (regression or bug-fix compared to Python 3.4) # - read1(), which fetches data before raising EOFError on next call # works here but is only available from Python>=3.3 +@deprecated('GzipFile.read1') def read1(gzf, size=-1): return gzf.read1(size) From 85e79ae792752353ea60bff3d93f9e77bea300fb Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 20 Aug 2019 21:09:49 +0500 Subject: [PATCH 097/113] Remove cStringIO imports. --- scrapy/downloadermiddlewares/decompression.py | 6 +----- scrapy/mail.py | 6 +----- scrapy/pipelines/files.py | 7 +------ scrapy/pipelines/images.py | 6 +----- scrapy/utils/gz.py | 9 ++------- scrapy/utils/iterators.py | 6 +----- tests/test_cmdline/__init__.py | 5 +---- tests/test_pipeline_media.py | 6 ++---- 8 files changed, 10 insertions(+), 41 deletions(-) diff --git a/scrapy/downloadermiddlewares/decompression.py b/scrapy/downloadermiddlewares/decompression.py index 49313cc04..e2d73f347 100644 --- a/scrapy/downloadermiddlewares/decompression.py +++ b/scrapy/downloadermiddlewares/decompression.py @@ -4,6 +4,7 @@ and extract the potentially compressed responses that may arrive. import bz2 import gzip +from io import BytesIO import zipfile import tarfile import logging @@ -11,11 +12,6 @@ from tempfile import mktemp import six -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO - from scrapy.responsetypes import responsetypes logger = logging.getLogger(__name__) diff --git a/scrapy/mail.py b/scrapy/mail.py index 746468e25..d24de2212 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -3,13 +3,9 @@ Mail sending helpers See documentation in docs/topics/email.rst """ +from io import BytesIO import logging -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO - from email.utils import COMMASPACE, formatdate from six.moves.email_mime_multipart import MIMEMultipart from six.moves.email_mime_text import MIMEText diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index cc3d10b63..8d74c5011 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -5,6 +5,7 @@ See documentation in topics/media-pipeline.rst """ import functools import hashlib +from io import BytesIO import mimetypes import os import os.path @@ -15,12 +16,6 @@ from six.moves.urllib.parse import urlparse from collections import defaultdict import six - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO - from twisted.internet import defer, threads from scrapy.pipelines.media import MediaPipeline diff --git a/scrapy/pipelines/images.py b/scrapy/pipelines/images.py index fa4d12ad1..e77cef4ff 100644 --- a/scrapy/pipelines/images.py +++ b/scrapy/pipelines/images.py @@ -5,13 +5,9 @@ See documentation in topics/media-pipeline.rst """ import functools import hashlib +from io import BytesIO import six -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO - from PIL import Image from scrapy.utils.misc import md5sum diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index dc8316d8c..f41e62fe3 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -1,12 +1,7 @@ -import struct - -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO from gzip import GzipFile - +from io import BytesIO import re +import struct from scrapy.utils.decorators import deprecated diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index dbc1e0d20..9693ba768 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -1,11 +1,7 @@ import re import csv -import logging -try: - from cStringIO import StringIO as BytesIO -except ImportError: - from io import BytesIO from io import StringIO +import logging import six from scrapy.http import TextResponse, Response diff --git a/tests/test_cmdline/__init__.py b/tests/test_cmdline/__init__.py index 68dfb1cca..56cfe642a 100644 --- a/tests/test_cmdline/__init__.py +++ b/tests/test_cmdline/__init__.py @@ -1,3 +1,4 @@ +from io import StringIO import json import os import pstats @@ -7,10 +8,6 @@ from subprocess import Popen, PIPE import sys import tempfile import unittest -try: - from cStringIO import StringIO -except ImportError: - from io import StringIO from scrapy.utils.test import get_testenv diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index 28e39cefa..ad2618ec9 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -144,10 +144,8 @@ class BaseMediaPipelineTestCase(unittest.TestCase): # The Failure should encapsulate a FileException ... self.assertEqual(failure.value, file_exc) - # ... and if we're running on Python 3 ... - if sys.version_info.major >= 3: - # ... it should have the returnValue exception set as its context - self.assertEqual(failure.value.__context__, def_gen_return_exc) + # ... and it should have the returnValue exception set as its context + self.assertEqual(failure.value.__context__, def_gen_return_exc) # Let's calculate the request fingerprint and fake some runtime data... fp = request_fingerprint(request) From cfa633f5e865fa491257215cc912d7321cc6da78 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 20 Aug 2019 21:13:01 +0500 Subject: [PATCH 098/113] Some text function messages cleanup, deprecate to_native_str. --- scrapy/http/response/__init__.py | 2 +- scrapy/utils/python.py | 8 ++++---- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index b0a526b72..a81404afb 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -88,7 +88,7 @@ class Response(object_ref): @property def text(self): """For subclasses of TextResponse, this will return the body - as text (unicode object in Python 2 and str in Python 3) + as str """ raise AttributeError("Response content isn't text") diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 37e6be868..663a8ebaa 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -90,7 +90,7 @@ def to_unicode(text, encoding=None, errors='strict'): if isinstance(text, six.text_type): return text if not isinstance(text, (bytes, six.text_type)): - raise TypeError('to_unicode must receive a bytes, str or unicode ' + raise TypeError('to_unicode must receive a bytes or str ' 'object, got %s' % type(text).__name__) if encoding is None: encoding = 'utf-8' @@ -103,16 +103,16 @@ def to_bytes(text, encoding=None, errors='strict'): if isinstance(text, bytes): return text if not isinstance(text, six.string_types): - raise TypeError('to_bytes must receive a unicode, str or bytes ' + raise TypeError('to_bytes must receive a str or bytes ' 'object, got %s' % type(text).__name__) if encoding is None: encoding = 'utf-8' return text.encode(encoding, errors) +@deprecated('to_unicode') def to_native_str(text, encoding=None, errors='strict'): - """ Return str representation of ``text`` - (bytes in Python 2.x and unicode in Python 3.x). """ + """ Return str representation of ``text``. """ return to_unicode(text, encoding, errors) From 92ffd2f249ec276436277385e525397a2fb8b0a0 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 20 Aug 2019 21:27:52 +0500 Subject: [PATCH 099/113] Simplify some more imports. --- scrapy/downloadermiddlewares/httpproxy.py | 5 +---- scrapy/loader/processors.py | 5 +---- tests/__init__.py | 5 ----- tests/test_downloader_handlers.py | 5 +---- tests/test_downloadermiddleware.py | 3 ++- tests/test_downloadermiddleware_robotstxt.py | 4 +++- tests/test_extension_telnet.py | 5 ----- tests/test_feedexport.py | 2 +- tests/test_http_request.py | 3 +-- tests/test_item.py | 2 +- tests/test_pipeline_files.py | 6 +----- tests/test_settings/__init__.py | 3 +-- tests/test_spider.py | 3 +-- tests/test_spidermiddleware.py | 3 ++- tests/test_stats.py | 6 +----- tests/test_utils_deprecate.py | 3 +-- tests/test_utils_misc/__init__.py | 2 +- tests/test_utils_trackref.py | 2 +- 18 files changed, 20 insertions(+), 47 deletions(-) diff --git a/scrapy/downloadermiddlewares/httpproxy.py b/scrapy/downloadermiddlewares/httpproxy.py index 2c35d1b90..2212d9688 100644 --- a/scrapy/downloadermiddlewares/httpproxy.py +++ b/scrapy/downloadermiddlewares/httpproxy.py @@ -1,10 +1,7 @@ import base64 from six.moves.urllib.parse import unquote, urlunparse from six.moves.urllib.request import getproxies, proxy_bypass -try: - from urllib2 import _parse_proxy -except ImportError: - from urllib.request import _parse_proxy +from urllib.request import _parse_proxy from scrapy.exceptions import NotConfigured from scrapy.utils.httpobj import urlparse_cached diff --git a/scrapy/loader/processors.py b/scrapy/loader/processors.py index 2acdc8093..02c625acc 100644 --- a/scrapy/loader/processors.py +++ b/scrapy/loader/processors.py @@ -3,10 +3,7 @@ This module provides some commonly used processors for Item Loaders. See documentation in docs/topics/loaders.rst """ -try: - from collections import ChainMap -except ImportError: - from scrapy.utils.datatypes import MergeDict as ChainMap +from collections import ChainMap from scrapy.utils.misc import arg_to_iter from scrapy.loader.common import wrap_loader_context diff --git a/tests/__init__.py b/tests/__init__.py index a54367f8c..12ce79fa9 100644 --- a/tests/__init__.py +++ b/tests/__init__.py @@ -21,11 +21,6 @@ if 'COV_CORE_CONFIG' in os.environ: os.environ['COV_CORE_CONFIG'] = os.path.join(_sourceroot, os.environ['COV_CORE_CONFIG']) -try: - import unittest.mock as mock -except ImportError: - import mock - tests_datadir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'sample_data') diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 109469503..6090998d4 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -2,11 +2,8 @@ import os import six import shutil import tempfile +from unittest import mock import contextlib -try: - from unittest import mock -except ImportError: - import mock from testfixtures import LogCapture from twisted.trial import unittest diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 03564e748..6b9a5bee8 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -1,3 +1,5 @@ +from unittest import mock + from twisted.trial.unittest import TestCase from twisted.python.failure import Failure @@ -7,7 +9,6 @@ from scrapy.exceptions import _InvalidOutput from scrapy.core.downloader.middleware import DownloaderMiddlewareManager from scrapy.utils.test import get_crawler from scrapy.utils.python import to_bytes -from tests import mock class ManagerTestCase(TestCase): diff --git a/tests/test_downloadermiddleware_robotstxt.py b/tests/test_downloadermiddleware_robotstxt.py index fbc46cba4..8266bf35f 100644 --- a/tests/test_downloadermiddleware_robotstxt.py +++ b/tests/test_downloadermiddleware_robotstxt.py @@ -1,5 +1,8 @@ # -*- coding: utf-8 -*- from __future__ import absolute_import + +from unittest import mock + from twisted.internet import reactor, error from twisted.internet.defer import Deferred, DeferredList, maybeDeferred from twisted.python import failure @@ -9,7 +12,6 @@ from scrapy.downloadermiddlewares.robotstxt import (RobotsTxtMiddleware, from scrapy.exceptions import IgnoreRequest, NotConfigured from scrapy.http import Request, Response, TextResponse from scrapy.settings import Settings -from tests import mock from tests.test_robotstxt_interface import rerp_available, reppy_available diff --git a/tests/test_extension_telnet.py b/tests/test_extension_telnet.py index 4f389e5cb..875ceb83c 100644 --- a/tests/test_extension_telnet.py +++ b/tests/test_extension_telnet.py @@ -1,8 +1,3 @@ -try: - import unittest.mock as mock -except ImportError: - import mock - from twisted.trial import unittest from twisted.conch.telnet import ITelnetProtocol from twisted.cred import credentials diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 11a5a8279..0c70bf80e 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -7,6 +7,7 @@ from io import BytesIO import tempfile import shutil import string +from unittest import mock from six.moves.urllib.parse import urljoin, urlparse, quote from six.moves.urllib.request import pathname2url @@ -15,7 +16,6 @@ from twisted.trial import unittest from twisted.internet import defer from scrapy.crawler import CrawlerRunner from scrapy.settings import Settings -from tests import mock from tests.mockserver import MockServer from w3lib.url import path_to_file_uri diff --git a/tests/test_http_request.py b/tests/test_http_request.py index bb451b5f4..9fe201579 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -3,6 +3,7 @@ import cgi import unittest import re import json +from unittest import mock from urllib.parse import unquote_to_bytes import warnings @@ -13,8 +14,6 @@ from six.moves.urllib.parse import urlparse, parse_qs, unquote from scrapy.http import Request, FormRequest, XmlRpcRequest, JsonRequest, Headers, HtmlResponse from scrapy.utils.python import to_bytes, to_native_str -from tests import mock - class RequestTest(unittest.TestCase): diff --git a/tests/test_item.py b/tests/test_item.py index 0ad278701..d98c63ddd 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -1,12 +1,12 @@ import sys import unittest +from unittest import mock from warnings import catch_warnings import six from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.item import ABCMeta, DictItem, Field, Item, ItemMeta -from tests import mock PY36_PLUS = (sys.version_info.major >= 3) and (sys.version_info.minor >= 6) diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index cb8f8da18..bd40e4103 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -1,10 +1,9 @@ import os import random import time -import hashlib -import warnings from tempfile import mkdtemp from shutil import rmtree +from unittest import mock from six.moves.urllib.parse import urlparse from six import BytesIO @@ -15,13 +14,10 @@ from scrapy.pipelines.files import FilesPipeline, FSFilesStore, S3FilesStore, GC from scrapy.item import Item, Field from scrapy.http import Request, Response from scrapy.settings import Settings -from scrapy.utils.python import to_bytes from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete from scrapy.utils.test import assert_gcs_environ, get_gcs_content_and_delete from scrapy.utils.boto import is_botocore -from tests import mock - def _mocked_download_func(request, info): response = request.meta.get('response') diff --git a/tests/test_settings/__init__.py b/tests/test_settings/__init__.py index 08286ff02..32e65bed5 100644 --- a/tests/test_settings/__init__.py +++ b/tests/test_settings/__init__.py @@ -1,10 +1,9 @@ import six import unittest -import warnings +from unittest import mock from scrapy.settings import (BaseSettings, Settings, SettingsAttribute, SETTINGS_PRIORITIES, get_settings_priority) -from tests import mock from . import default_settings diff --git a/tests/test_spider.py b/tests/test_spider.py index 6f6cdb8ff..c0fccfdd6 100644 --- a/tests/test_spider.py +++ b/tests/test_spider.py @@ -1,5 +1,6 @@ import gzip import inspect +from unittest import mock import warnings from io import BytesIO @@ -17,8 +18,6 @@ from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.trackref import object_ref from scrapy.utils.test import get_crawler -from tests import mock - class SpiderTest(unittest.TestCase): diff --git a/tests/test_spidermiddleware.py b/tests/test_spidermiddleware.py index 832fd3330..55d665e79 100644 --- a/tests/test_spidermiddleware.py +++ b/tests/test_spidermiddleware.py @@ -1,3 +1,5 @@ +from unittest import mock + from twisted.trial.unittest import TestCase from twisted.python.failure import Failure @@ -6,7 +8,6 @@ from scrapy.http import Request, Response from scrapy.exceptions import _InvalidOutput from scrapy.utils.test import get_crawler from scrapy.core.spidermw import SpiderMiddlewareManager -from tests import mock class SpiderMiddlewareTestCase(TestCase): diff --git a/tests/test_stats.py b/tests/test_stats.py index 2033dbe07..2bbbb9e2c 100644 --- a/tests/test_stats.py +++ b/tests/test_stats.py @@ -1,10 +1,6 @@ from datetime import datetime import unittest - -try: - from unittest import mock -except ImportError: - import mock +from unittest import mock from scrapy.extensions.corestats import CoreStats from scrapy.spiders import Spider diff --git a/tests/test_utils_deprecate.py b/tests/test_utils_deprecate.py index 3e7236fb1..ce04e7f29 100644 --- a/tests/test_utils_deprecate.py +++ b/tests/test_utils_deprecate.py @@ -2,12 +2,11 @@ from __future__ import absolute_import import inspect import unittest +from unittest import mock import warnings from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.deprecate import create_deprecated_class, update_classpath -from tests import mock - class MyWarning(UserWarning): pass diff --git a/tests/test_utils_misc/__init__.py b/tests/test_utils_misc/__init__.py index e109d5343..de9da9104 100644 --- a/tests/test_utils_misc/__init__.py +++ b/tests/test_utils_misc/__init__.py @@ -1,11 +1,11 @@ import sys import os import unittest +from unittest import mock from scrapy.item import Item, Field from scrapy.utils.misc import arg_to_iter, create_instance, load_object, set_environ, walk_modules -from tests import mock __doctests__ = ['scrapy.utils.misc'] diff --git a/tests/test_utils_trackref.py b/tests/test_utils_trackref.py index c6072fc0d..480a717e7 100644 --- a/tests/test_utils_trackref.py +++ b/tests/test_utils_trackref.py @@ -1,7 +1,7 @@ import six import unittest +from unittest import mock from scrapy.utils import trackref -from tests import mock class Foo(trackref.object_ref): From a138fb05d4f0d90e2002e85a348a5be34904d3d8 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 20 Aug 2019 21:35:13 +0500 Subject: [PATCH 100/113] Replace to_native_str calls with to_unicode. --- scrapy/core/downloader/handlers/http11.py | 2 +- scrapy/downloadermiddlewares/cookies.py | 6 +++--- scrapy/downloadermiddlewares/robotstxt.py | 3 --- scrapy/exporters.py | 4 ++-- scrapy/http/cookies.py | 10 +++++----- scrapy/http/response/text.py | 8 ++++---- scrapy/linkextractors/lxmlhtml.py | 4 ++-- scrapy/responsetypes.py | 6 +++--- scrapy/robotstxt.py | 8 ++++---- scrapy/spidermiddlewares/referer.py | 5 ++--- scrapy/utils/reqser.py | 4 ++-- scrapy/utils/request.py | 4 ++-- scrapy/utils/response.py | 4 ++-- scrapy/utils/ssl.py | 4 ++-- tests/test_command_parse.py | 5 ++--- tests/test_commands.py | 7 ++----- tests/test_feedexport.py | 7 +++---- tests/test_http_request.py | 7 +++---- tests/test_http_response.py | 6 +++--- tests/test_robotstxt_interface.py | 1 - 20 files changed, 47 insertions(+), 58 deletions(-) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 91b45a8fc..7d917cb74 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -174,7 +174,7 @@ def tunnel_request_data(host, port, proxy_auth_header=None): r""" Return binary content of a CONNECT request. - >>> from scrapy.utils.python import to_native_str as s + >>> from scrapy.utils.python import to_unicode as s >>> s(tunnel_request_data("example.com", 8080)) 'CONNECT example.com:8080 HTTP/1.1\r\nHost: example.com:8080\r\n\r\n' >>> s(tunnel_request_data("example.com", 8080, b"123")) diff --git a/scrapy/downloadermiddlewares/cookies.py b/scrapy/downloadermiddlewares/cookies.py index 321c0171b..0d2b9900c 100644 --- a/scrapy/downloadermiddlewares/cookies.py +++ b/scrapy/downloadermiddlewares/cookies.py @@ -6,7 +6,7 @@ from collections import defaultdict from scrapy.exceptions import NotConfigured from scrapy.http import Response from scrapy.http.cookies import CookieJar -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode logger = logging.getLogger(__name__) @@ -53,7 +53,7 @@ class CookiesMiddleware(object): def _debug_cookie(self, request, spider): if self.debug: - cl = [to_native_str(c, errors='replace') + cl = [to_unicode(c, errors='replace') for c in request.headers.getlist('Cookie')] if cl: cookies = "\n".join("Cookie: {}\n".format(c) for c in cl) @@ -62,7 +62,7 @@ class CookiesMiddleware(object): def _debug_set_cookie(self, response, spider): if self.debug: - cl = [to_native_str(c, errors='replace') + cl = [to_unicode(c, errors='replace') for c in response.headers.getlist('Set-Cookie')] if cl: cookies = "\n".join("Set-Cookie: {}\n".format(c) for c in cl) diff --git a/scrapy/downloadermiddlewares/robotstxt.py b/scrapy/downloadermiddlewares/robotstxt.py index 6a5dfb79c..251706c50 100644 --- a/scrapy/downloadermiddlewares/robotstxt.py +++ b/scrapy/downloadermiddlewares/robotstxt.py @@ -5,15 +5,12 @@ enable this middleware and enable the ROBOTSTXT_OBEY setting. """ import logging -import sys -import re from twisted.internet.defer import Deferred, maybeDeferred from scrapy.exceptions import NotConfigured, IgnoreRequest from scrapy.http import Request from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.log import failure_to_exc_info -from scrapy.utils.python import to_native_str from scrapy.utils.misc import load_object logger = logging.getLogger(__name__) diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 1aa195b0b..8eb52995e 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -12,7 +12,7 @@ from six.moves import cPickle as pickle from xml.sax.saxutils import XMLGenerator from scrapy.utils.serialize import ScrapyJSONEncoder -from scrapy.utils.python import to_bytes, to_unicode, to_native_str, is_listlike +from scrapy.utils.python import to_bytes, to_unicode, is_listlike from scrapy.item import BaseItem from scrapy.exceptions import ScrapyDeprecationWarning import warnings @@ -232,7 +232,7 @@ class CsvItemExporter(BaseItemExporter): def _build_row(self, values): for s in values: try: - yield to_native_str(s, self.encoding) + yield to_unicode(s, self.encoding) except TypeError: yield s diff --git a/scrapy/http/cookies.py b/scrapy/http/cookies.py index 4e8056750..4532c3ab7 100644 --- a/scrapy/http/cookies.py +++ b/scrapy/http/cookies.py @@ -3,7 +3,7 @@ from six.moves.http_cookiejar import ( CookieJar as _CookieJar, DefaultCookiePolicy, IPV4_RE ) from scrapy.utils.httpobj import urlparse_cached -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode class CookieJar(object): @@ -165,13 +165,13 @@ class WrappedRequest(object): return name in self.request.headers def get_header(self, name, default=None): - return to_native_str(self.request.headers.get(name, default), + return to_unicode(self.request.headers.get(name, default), errors='replace') def header_items(self): return [ - (to_native_str(k, errors='replace'), - [to_native_str(x, errors='replace') for x in v]) + (to_unicode(k, errors='replace'), + [to_unicode(x, errors='replace') for x in v]) for k, v in self.request.headers.items() ] @@ -189,7 +189,7 @@ class WrappedResponse(object): # python3 cookiejars calls get_all def get_all(self, name, default=None): - return [to_native_str(v, errors='replace') + return [to_unicode(v, errors='replace') for v in self.response.headers.getlist(name)] # python2 cookiejars calls getheaders getheaders = get_all diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index a8010877c..37f450e54 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -16,7 +16,7 @@ from w3lib.html import strip_html5_whitespace from scrapy.http.request import Request from scrapy.http.response import Response from scrapy.utils.response import get_base_url -from scrapy.utils.python import memoizemethod_noargs, to_native_str +from scrapy.utils.python import memoizemethod_noargs, to_unicode class TextResponse(Response): @@ -32,7 +32,7 @@ class TextResponse(Response): def _set_url(self, url): if isinstance(url, six.text_type): - self._url = to_native_str(url, self.encoding) + self._url = to_unicode(url, self.encoding) else: super(TextResponse, self)._set_url(url) @@ -81,11 +81,11 @@ class TextResponse(Response): @memoizemethod_noargs def _headers_encoding(self): content_type = self.headers.get(b'Content-Type', b'') - return http_content_type_encoding(to_native_str(content_type)) + return http_content_type_encoding(to_unicode(content_type)) def _body_inferred_encoding(self): if self._cached_benc is None: - content_type = to_native_str(self.headers.get(b'Content-Type', b'')) + content_type = to_unicode(self.headers.get(b'Content-Type', b'')) benc, ubody = html_to_unicode(content_type, self.body, auto_detect_fun=self._auto_detect_fun, default_encoding=self._DEFAULT_ENCODING) diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index 8f6f93a44..890c019c8 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -10,7 +10,7 @@ from w3lib.url import canonicalize_url from scrapy.link import Link from scrapy.utils.misc import arg_to_iter, rel_has_nofollow -from scrapy.utils.python import unique as unique_list, to_native_str +from scrapy.utils.python import unique as unique_list, to_unicode from scrapy.utils.response import get_base_url from scrapy.linkextractors import FilteringLinkExtractor @@ -67,7 +67,7 @@ class LxmlParserLinkExtractor(object): url = self.process_attr(attr_val) if url is None: continue - url = to_native_str(url, encoding=response_encoding) + url = to_unicode(url, encoding=response_encoding) # to fix relative links after process_value url = urljoin(response_url, url) link = Link(url, _collect_string_content(el) or u'', diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 4a2d5bf52..de62276c8 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -10,7 +10,7 @@ import six from scrapy.http import Response from scrapy.utils.misc import load_object -from scrapy.utils.python import binary_is_text, to_bytes, to_native_str +from scrapy.utils.python import binary_is_text, to_bytes, to_unicode class ResponseTypes(object): @@ -55,12 +55,12 @@ class ResponseTypes(object): header """ if content_encoding: return Response - mimetype = to_native_str(content_type).split(';')[0].strip().lower() + mimetype = to_unicode(content_type).split(';')[0].strip().lower() return self.from_mimetype(mimetype) def from_content_disposition(self, content_disposition): try: - filename = to_native_str(content_disposition, + filename = to_unicode(content_disposition, encoding='latin-1', errors='replace').split(';')[1].split('=')[1] filename = filename.strip('"\'') return self.from_filename(filename) diff --git a/scrapy/robotstxt.py b/scrapy/robotstxt.py index 189f165d1..95a8c09b8 100644 --- a/scrapy/robotstxt.py +++ b/scrapy/robotstxt.py @@ -3,14 +3,14 @@ import logging from abc import ABCMeta, abstractmethod from six import with_metaclass -from scrapy.utils.python import to_native_str, to_unicode +from scrapy.utils.python import to_unicode logger = logging.getLogger(__name__) def decode_robotstxt(robotstxt_body, spider, to_native_str_type=False): try: if to_native_str_type: - robotstxt_body = to_native_str(robotstxt_body) + robotstxt_body = to_unicode(robotstxt_body) else: robotstxt_body = robotstxt_body.decode('utf-8') except UnicodeDecodeError: @@ -66,8 +66,8 @@ class PythonRobotParser(RobotParser): return o def allowed(self, url, user_agent): - user_agent = to_native_str(user_agent) - url = to_native_str(url) + user_agent = to_unicode(user_agent) + url = to_unicode(url) return self.rp.can_fetch(user_agent, url) diff --git a/scrapy/spidermiddlewares/referer.py b/scrapy/spidermiddlewares/referer.py index 1ddfb37f4..c76e4d5a2 100644 --- a/scrapy/spidermiddlewares/referer.py +++ b/scrapy/spidermiddlewares/referer.py @@ -10,8 +10,7 @@ from w3lib.url import safe_url_string from scrapy.http import Request, Response from scrapy.exceptions import NotConfigured from scrapy import signals -from scrapy.utils.python import to_native_str -from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.python import to_unicode from scrapy.utils.misc import load_object from scrapy.utils.url import strip_url @@ -322,7 +321,7 @@ class RefererMiddleware(object): if isinstance(resp_or_url, Response): policy_header = resp_or_url.headers.get('Referrer-Policy') if policy_header is not None: - policy_name = to_native_str(policy_header.decode('latin1')) + policy_name = to_unicode(policy_header.decode('latin1')) if policy_name is None: return self.default_policy() diff --git a/scrapy/utils/reqser.py b/scrapy/utils/reqser.py index c7ea7b425..495564ac0 100644 --- a/scrapy/utils/reqser.py +++ b/scrapy/utils/reqser.py @@ -4,7 +4,7 @@ Helper functions for serializing (and deserializing) requests. import six from scrapy.http import Request -from scrapy.utils.python import to_unicode, to_native_str +from scrapy.utils.python import to_unicode from scrapy.utils.misc import load_object @@ -54,7 +54,7 @@ def request_from_dict(d, spider=None): eb = _get_method(spider, eb) request_cls = load_object(d['_class']) if '_class' in d else Request return request_cls( - url=to_native_str(d['url']), + url=to_unicode(d['url']), callback=cb, errback=eb, method=d['method'], diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index fb5af66a2..63d0ae772 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -9,7 +9,7 @@ import weakref from six.moves.urllib.parse import urlunparse from w3lib.http import basic_auth_header -from scrapy.utils.python import to_bytes, to_native_str +from scrapy.utils.python import to_bytes, to_unicode from w3lib.url import canonicalize_url from scrapy.utils.httpobj import urlparse_cached @@ -97,4 +97,4 @@ def referer_str(request): referrer = request.headers.get('Referer') if referrer is None: return referrer - return to_native_str(referrer, errors='replace') + return to_unicode(referrer, errors='replace') diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index c3236afd4..feab07431 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -8,7 +8,7 @@ import webbrowser import tempfile from twisted.web import http -from scrapy.utils.python import to_bytes, to_native_str +from scrapy.utils.python import to_bytes, to_unicode from w3lib import html @@ -36,7 +36,7 @@ def response_status_message(status): """Return status code plus status text descriptive message """ message = http.RESPONSES.get(int(status), "Unknown Status") - return '%s %s' % (status, to_native_str(message)) + return '%s %s' % (status, to_unicode(message)) def response_httprepr(response): diff --git a/scrapy/utils/ssl.py b/scrapy/utils/ssl.py index 02aed60ee..6e81b33ff 100644 --- a/scrapy/utils/ssl.py +++ b/scrapy/utils/ssl.py @@ -3,7 +3,7 @@ import OpenSSL import OpenSSL._util as pyOpenSSLutil -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode # The OpenSSL symbol is present since 1.1.1 but it's not currently supported in any version of pyOpenSSL. @@ -12,7 +12,7 @@ SSL_OP_NO_TLSv1_3 = getattr(pyOpenSSLutil.lib, 'SSL_OP_NO_TLSv1_3', 0) def ffi_buf_to_string(buf): - return to_native_str(pyOpenSSLutil.ffi.string(buf)) + return to_unicode(pyOpenSSLutil.ffi.string(buf)) def x509name_to_string(x509name): diff --git a/tests/test_command_parse.py b/tests/test_command_parse.py index 62d5d76b4..b134beb88 100644 --- a/tests/test_command_parse.py +++ b/tests/test_command_parse.py @@ -1,17 +1,16 @@ import os from os.path import join, abspath -from twisted.trial import unittest from twisted.internet import defer from scrapy.utils.testsite import SiteTest from scrapy.utils.testproc import ProcessTest -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode from tests.test_commands import CommandTest def _textmode(bstr): """Normalize input the same as writing to a file and reading from it in text mode""" - return to_native_str(bstr).replace(os.linesep, '\n') + return to_unicode(bstr).replace(os.linesep, '\n') class ParseCommandTest(ProcessTest, SiteTest, CommandTest): command = 'parse' diff --git a/tests/test_commands.py b/tests/test_commands.py index b8445ae6c..536379170 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -10,13 +10,10 @@ from contextlib import contextmanager from threading import Timer from twisted.trial import unittest -from twisted.internet import defer import scrapy -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode from scrapy.utils.test import get_testenv -from scrapy.utils.testsite import SiteTest -from scrapy.utils.testproc import ProcessTest from tests.test_crawler import ExceptionSpider, NoRequestsSpider @@ -56,7 +53,7 @@ class ProjectTest(unittest.TestCase): finally: timer.cancel() - return p, to_native_str(stdout), to_native_str(stderr) + return p, to_unicode(stdout), to_unicode(stderr) class StartprojectTest(ProjectTest): diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 0c70bf80e..87139e81f 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -26,8 +26,7 @@ from scrapy.extensions.feedexport import ( S3FeedStorage, StdoutFeedStorage, BlockingFeedStorage) from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get_crawler -from scrapy.utils.python import to_native_str -from scrapy.utils.project import get_project_settings +from scrapy.utils.python import to_unicode from pathlib import Path @@ -459,7 +458,7 @@ class FeedExportTest(unittest.TestCase): settings.update({'FEED_FORMAT': 'csv'}) data = yield self.exported_data(items, settings) - reader = csv.DictReader(to_native_str(data).splitlines()) + reader = csv.DictReader(to_unicode(data).splitlines()) got_rows = list(reader) if ordered: self.assertEqual(reader.fieldnames, header) @@ -473,7 +472,7 @@ class FeedExportTest(unittest.TestCase): settings = settings or {} settings.update({'FEED_FORMAT': 'jl'}) data = yield self.exported_data(items, settings) - parsed = [json.loads(to_native_str(line)) for line in data.splitlines()] + parsed = [json.loads(to_unicode(line)) for line in data.splitlines()] rows = [{k: v for k, v in row.items() if v} for row in rows] self.assertEqual(rows, parsed) diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 9fe201579..3518da21c 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -7,12 +7,11 @@ from unittest import mock from urllib.parse import unquote_to_bytes import warnings -import six from six.moves import xmlrpc_client as xmlrpclib from six.moves.urllib.parse import urlparse, parse_qs, unquote from scrapy.http import Request, FormRequest, XmlRpcRequest, JsonRequest, Headers, HtmlResponse -from scrapy.utils.python import to_bytes, to_native_str +from scrapy.utils.python import to_bytes, to_unicode class RequestTest(unittest.TestCase): @@ -349,8 +348,8 @@ class FormRequestTest(RequestTest): request_class = FormRequest def assertQueryEqual(self, first, second, msg=None): - first = to_native_str(first).split("&") - second = to_native_str(second).split("&") + first = to_unicode(first).split("&") + second = to_unicode(second).split("&") return self.assertEqual(sorted(first), sorted(second), msg) def test_empty_formdata(self): diff --git a/tests/test_http_response.py b/tests/test_http_response.py index ec3b51086..883c943da 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -7,7 +7,7 @@ from w3lib.encoding import resolve_encoding from scrapy.http import (Request, Response, TextResponse, HtmlResponse, XmlResponse, Headers) from scrapy.selector import Selector -from scrapy.utils.python import to_native_str +from scrapy.utils.python import to_unicode from scrapy.exceptions import NotSupported from scrapy.link import Link from tests import get_testdata @@ -204,11 +204,11 @@ class TextResponseTest(BaseResponseTest): assert isinstance(resp.url, str) resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='utf-8') - self.assertEqual(resp.url, to_native_str(b'http://www.example.com/price/\xc2\xa3')) + self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3')) resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='latin-1') self.assertEqual(resp.url, 'http://www.example.com/price/\xa3') resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=utf-8"]}) - self.assertEqual(resp.url, to_native_str(b'http://www.example.com/price/\xc2\xa3')) + self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3')) resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=iso-8859-1"]}) self.assertEqual(resp.url, 'http://www.example.com/price/\xa3') diff --git a/tests/test_robotstxt_interface.py b/tests/test_robotstxt_interface.py index 9aaab560a..cd7480e33 100644 --- a/tests/test_robotstxt_interface.py +++ b/tests/test_robotstxt_interface.py @@ -1,6 +1,5 @@ # coding=utf-8 from twisted.trial import unittest -from scrapy.utils.python import to_native_str def reppy_available(): From 87c23ba22d2ef714d778b62c794333acaf232f60 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 28 Aug 2019 16:29:53 +0500 Subject: [PATCH 101/113] Remove Py2-only code that checks sys.version_info. --- tests/test_utils_reqser.py | 7 ++----- 1 file changed, 2 insertions(+), 5 deletions(-) diff --git a/tests/test_utils_reqser.py b/tests/test_utils_reqser.py index 11ac56897..92cd16de7 100644 --- a/tests/test_utils_reqser.py +++ b/tests/test_utils_reqser.py @@ -80,8 +80,6 @@ class RequestSerializationTest(unittest.TestCase): self._assert_serializes_ok(r, spider=self.spider) def test_mixin_private_callback_serialization(self): - if sys.version_info[0] < 3: - return r = Request("http://www.example.com", callback=self.spider._TestSpiderMixin__mixin_callback, errback=self.spider.handle_error) @@ -119,9 +117,8 @@ class RequestSerializationTest(unittest.TestCase): def test_private_name_mangling(self): self._assert_mangles_to( self.spider, '_TestSpider__parse_item_private') - if sys.version_info[0] >= 3: - self._assert_mangles_to( - self.spider, '_TestSpiderMixin__mixin_callback') + self._assert_mangles_to( + self.spider, '_TestSpiderMixin__mixin_callback') def test_unserializable_callback1(self): r = Request("http://www.example.com", callback=lambda x: x) From a9c891399d1bf8a888392afe80c82a8ec8f2e8e7 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 31 Oct 2019 22:55:58 +0500 Subject: [PATCH 102/113] Fix a duplicate ref name in docs. --- docs/topics/downloader-middleware.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 366b95510..e93645077 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -474,7 +474,7 @@ DBM storage backend A DBM_ storage backend is also available for the HTTP cache middleware. - By default, it uses the dbm_ module, but you can change it with the + By default, it uses the `dbm module`_, but you can change it with the :setting:`HTTPCACHE_DBM_MODULE` setting. .. _httpcache-storage-custom: @@ -1202,4 +1202,4 @@ The default encoding for proxy authentication on :class:`HttpProxyMiddleware`. .. _DBM: https://en.wikipedia.org/wiki/Dbm -.. _dbm: https://docs.python.org/3/library/dbm.html +.. _dbm module: https://docs.python.org/3/library/dbm.html From dd367438fa7a7fec923b28648c4e909cbed1b47d Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Fri, 1 Nov 2019 20:05:37 +0500 Subject: [PATCH 103/113] Improve the dbm module ref. MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- docs/topics/downloader-middleware.rst | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index e93645077..ae6d41809 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -474,7 +474,7 @@ DBM storage backend A DBM_ storage backend is also available for the HTTP cache middleware. - By default, it uses the `dbm module`_, but you can change it with the + By default, it uses the :mod:`dbm`, but you can change it with the :setting:`HTTPCACHE_DBM_MODULE` setting. .. _httpcache-storage-custom: @@ -1202,4 +1202,3 @@ The default encoding for proxy authentication on :class:`HttpProxyMiddleware`. .. _DBM: https://en.wikipedia.org/wiki/Dbm -.. _dbm module: https://docs.python.org/3/library/dbm.html From be6da52019990c1db45b1101dd99787752a14313 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 14 Nov 2019 10:31:55 +0100 Subject: [PATCH 104/113] Include extensions from #2067 --- scrapy/linkextractors/__init__.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index 3c75e683d..a2ac963fe 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -19,9 +19,12 @@ from scrapy.utils.url import ( # common file extensions that are not followed if they occur in links IGNORED_EXTENSIONS = [ + # archives + '7z', '7zip', 'bz2', 'rar', 'tar', 'tar.gz', 'xz', 'zip', + # images 'mng', 'pct', 'bmp', 'gif', 'jpg', 'jpeg', 'png', 'pst', 'psp', 'tif', - 'tiff', 'ai', 'drw', 'dxf', 'eps', 'ps', 'svg', + 'tiff', 'ai', 'drw', 'dxf', 'eps', 'ps', 'svg', 'cdr', 'ico', # audio 'mp3', 'wma', 'ogg', 'wav', 'ra', 'aac', 'mid', 'au', 'aiff', @@ -35,7 +38,7 @@ IGNORED_EXTENSIONS = [ 'odp', # other - 'css', 'pdf', 'exe', 'bin', 'rss', 'zip', 'rar', 'dmg', 'iso', 'apk', 'xz' + 'css', 'pdf', 'exe', 'bin', 'rss', 'dmg', 'iso', 'apk' ] From 3631453bfb1fb2426916919dbfd489ba9ffd9505 Mon Sep 17 00:00:00 2001 From: Andrey Rahmatullin Date: Thu, 14 Nov 2019 15:07:53 +0500 Subject: [PATCH 105/113] Remove spaces on a blank line. --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index a2ac963fe..e4c62f87b 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -21,7 +21,7 @@ from scrapy.utils.url import ( IGNORED_EXTENSIONS = [ # archives '7z', '7zip', 'bz2', 'rar', 'tar', 'tar.gz', 'xz', 'zip', - + # images 'mng', 'pct', 'bmp', 'gif', 'jpg', 'jpeg', 'png', 'pst', 'psp', 'tif', 'tiff', 'ai', 'drw', 'dxf', 'eps', 'ps', 'svg', 'cdr', 'ico', From e291460db67ef8c9e52de02a0a4a86f4bce39b9d Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 14 Nov 2019 15:24:37 +0500 Subject: [PATCH 106/113] Fix flake8-detected errors. --- scrapy/crawler.py | 1 - scrapy/exporters.py | 1 - scrapy/http/cookies.py | 2 +- scrapy/link.py | 2 ++ tests/test_pipeline_media.py | 2 -- 5 files changed, 3 insertions(+), 5 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 19b998e0d..8868a985b 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -3,7 +3,6 @@ import signal import logging import warnings -import sys from twisted.internet import reactor, defer from zope.interface.verify import verifyClass, DoesNotImplement diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 8eb52995e..3defafd60 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -4,7 +4,6 @@ Item Exporters are used to export/serialize items into different formats. import csv import io -import sys import pprint import marshal import six diff --git a/scrapy/http/cookies.py b/scrapy/http/cookies.py index 4532c3ab7..60a14c6f8 100644 --- a/scrapy/http/cookies.py +++ b/scrapy/http/cookies.py @@ -166,7 +166,7 @@ class WrappedRequest(object): def get_header(self, name, default=None): return to_unicode(self.request.headers.get(name, default), - errors='replace') + errors='replace') def header_items(self): return [ diff --git a/scrapy/link.py b/scrapy/link.py index be1888ef0..a809c5ca4 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -4,6 +4,8 @@ This module defines the Link object used in Link extractors. For actual link extractors implementation see scrapy.linkextractors, or its documentation in: docs/topics/link-extractors.rst """ + + class Link(object): """Link objects represent an extracted link by the LinkExtractor.""" diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index ad2618ec9..ad958e25f 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -1,7 +1,5 @@ from __future__ import print_function -import sys - from testfixtures import LogCapture from twisted.trial import unittest from twisted.python.failure import Failure From b8ef12cd4707f9e095abdd26cc137558009c335f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 14 Nov 2019 12:10:25 +0100 Subject: [PATCH 107/113] Add bandit to CI --- .bandit.yml | 16 ++++++++++++++++ .travis.yml | 2 ++ tox.ini | 7 +++++++ 3 files changed, 25 insertions(+) create mode 100644 .bandit.yml diff --git a/.bandit.yml b/.bandit.yml new file mode 100644 index 000000000..00554587a --- /dev/null +++ b/.bandit.yml @@ -0,0 +1,16 @@ +skips: +- B101 +- B105 +- B303 +- B306 +- B307 +- B311 +- B320 +- B321 +- B402 +- B404 +- B406 +- B410 +- B503 +- B603 +- B605 diff --git a/.travis.yml b/.travis.yml index 0e77af9fd..9f477e860 100644 --- a/.travis.yml +++ b/.travis.yml @@ -7,6 +7,8 @@ branches: - /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/ matrix: include: + - env: TOXENV=security + python: 3.8 - env: TOXENV=flake8 python: 3.8 - env: TOXENV=pypy3 diff --git a/tox.ini b/tox.ini index 3668058c3..cd575f3c5 100644 --- a/tox.ini +++ b/tox.ini @@ -62,6 +62,13 @@ basepython = pypy3 commands = py.test {posargs:docs scrapy tests} +[testenv:security] +basepython = python3.8 +deps = + bandit +commands = + bandit -r -c .bandit.yml {posargs:scrapy} + [testenv:flake8] basepython = python3.8 deps = From 5ee5508cc33116b3700c9b745138b56cc67951f7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 14 Nov 2019 15:42:34 +0100 Subject: [PATCH 108/113] Have CI record the 10 slowest tests --- tox.ini | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tox.ini b/tox.ini index 3668058c3..ec04035f5 100644 --- a/tox.ini +++ b/tox.ini @@ -21,7 +21,7 @@ passenv = GCS_TEST_FILE_URI GCS_PROJECT_ID commands = - py.test --cov=scrapy --cov-report= {posargs:docs scrapy tests} + py.test --cov=scrapy --cov-report= {posargs:--durations=10 docs scrapy tests} [testenv:py35] basepython = python3.5 @@ -60,7 +60,7 @@ basepython = python3.8 [testenv:pypy3] basepython = pypy3 commands = - py.test {posargs:docs scrapy tests} + py.test {posargs:--durations=10 docs scrapy tests} [testenv:flake8] basepython = python3.8 From fe3a121f1358fc904915f5b32e276520523c553a Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 14 Nov 2019 22:50:53 +0500 Subject: [PATCH 109/113] Use kwargs when calling get_func_args. --- tests/test_utils_python.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index 096aa50b7..a94398796 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -232,10 +232,10 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertEqual(get_func_args(operator.itemgetter(2)), []) else: self.assertEqual( - get_func_args(six.text_type.split, True), ['sep', 'maxsplit']) - self.assertEqual(get_func_args(" ".join, True), ['list']) + get_func_args(six.text_type.split, stripself=True), ['sep', 'maxsplit']) + self.assertEqual(get_func_args(" ".join, stripself=True), ['list']) self.assertEqual( - get_func_args(operator.itemgetter(2), True), ['obj']) + get_func_args(operator.itemgetter(2), stripself=True), ['obj']) def test_without_none_values(self): From 3b2289ad012043b94b495d411316b2778bf3db35 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 14 Nov 2019 22:53:28 +0500 Subject: [PATCH 110/113] Rename test_non_str_url_py2 to test_bytes_url. --- tests/test_link.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_link.py b/tests/test_link.py index 5e2ce5eeb..e0f1efffa 100644 --- a/tests/test_link.py +++ b/tests/test_link.py @@ -43,6 +43,6 @@ class LinkTest(unittest.TestCase): l2 = eval(repr(l1)) self._assert_same_links(l1, l2) - def test_non_str_url_py2(self): + def test_bytes_url(self): with self.assertRaises(TypeError): Link(b"http://www.example.com/\xc2\xa3") From 77a84f620ffa5d001072c7122c0f38048fe15606 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Jos=C3=A9=20Alberto=20/=20Speedy?= Date: Fri, 15 Nov 2019 11:09:24 +0100 Subject: [PATCH 111/113] Fix string MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- README.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/README.rst b/README.rst index 87eaac2af..4c038030f 100644 --- a/README.rst +++ b/README.rst @@ -62,7 +62,7 @@ directory. Releases ======== -You can check https://docs.scrapy.org/en/latest/news.html for release notes. +You can check https://docs.scrapy.org/en/latest/news.html for the release notes. Community (blog, twitter, mail list, IRC) ========================================= From 0e252f5a13be3195fdc3ec1d66a111ae01a0ab80 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot Date: Fri, 15 Nov 2019 19:12:43 +0100 Subject: [PATCH 112/113] fix E711 and E713 --- pytest.ini | 4 ++-- tests/test_downloader_handlers.py | 4 ++-- tests/test_downloadermiddleware_httpcache.py | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/pytest.ini b/pytest.ini index fa6e7287e..529ad5d27 100644 --- a/pytest.ini +++ b/pytest.ini @@ -192,14 +192,14 @@ flake8-ignore = tests/test_crawl.py E501 E741 E265 tests/test_crawler.py F841 E306 E501 tests/test_dependencies.py E302 F841 E501 E305 - tests/test_downloader_handlers.py E124 E127 E128 E225 E261 E265 F401 E501 E502 E701 E711 E126 E226 E123 + tests/test_downloader_handlers.py E124 E127 E128 E225 E261 E265 F401 E501 E502 E701 E126 E226 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E302 E501 tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126 tests/test_downloadermiddleware_decompression.py E127 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 - tests/test_downloadermiddleware_httpcache.py E713 E501 E302 E305 F401 + tests/test_downloadermiddleware_httpcache.py E501 E302 E305 F401 tests/test_downloadermiddleware_httpcompression.py E501 F401 E251 E126 E123 tests/test_downloadermiddleware_httpproxy.py F401 E501 E128 tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 6090998d4..59d4a3eec 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -615,7 +615,7 @@ class Http11MockServerTestCase(unittest.TestCase): crawler = get_crawler(SingleRequestSpider) yield crawler.crawl(seed=Request(url=self.mockserver.url(''))) failure = crawler.spider.meta.get('failure') - self.assertTrue(failure == None) + self.assertTrue(failure is None) reason = crawler.spider.meta['close_reason'] self.assertTrue(reason, 'finished') @@ -636,7 +636,7 @@ class Http11MockServerTestCase(unittest.TestCase): yield crawler.crawl(seed=request) # download_maxsize = 50 is enough for the gzipped response failure = crawler.spider.meta.get('failure') - self.assertTrue(failure == None) + self.assertTrue(failure is None) reason = crawler.spider.meta['close_reason'] self.assertTrue(reason, 'finished') else: diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 950664ffe..9d863b6e3 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -85,8 +85,8 @@ class _BaseTest(unittest.TestCase): def assertEqualRequestButWithCacheValidators(self, request1, request2): self.assertEqual(request1.url, request2.url) - assert not b'If-None-Match' in request1.headers - assert not b'If-Modified-Since' in request1.headers + assert b'If-None-Match' not in request1.headers + assert b'If-Modified-Since' not in request1.headers assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) From 78ad01632f16a97fb180d8c2972075b19f471380 Mon Sep 17 00:00:00 2001 From: Andrey Rahmatullin Date: Tue, 19 Nov 2019 14:43:30 +0500 Subject: [PATCH 113/113] Fix flake8 problems in PR #3989 (#4176) --- tests/test_logformatter.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index 5b5d68f4f..afbd25d0c 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -62,7 +62,7 @@ class LogFormatterTestCase(unittest.TestCase): lines = logline.splitlines() assert all(isinstance(x, six.text_type) for x in lines) self.assertEqual(lines, [u"Dropped: \u2018", '{}']) - + def test_error(self): # In practice, the complete traceback is shown by passing the # 'exc_info' argument to the logging function @@ -121,7 +121,7 @@ class LogformatterSubclassTest(LogFormatterTestCase): "Crawled (200) (referer: http://example.com) ['cached']") def test_flags_in_request(self): - req = Request("http://www.example.com", flags=['test','flag']) + req = Request("http://www.example.com", flags=['test', 'flag']) res = Response("http://www.example.com") logkws = self.formatter.crawled(req, res, self.spider) logline = logkws['msg'] % logkws['args']