From 1b591ff061f2b38bf328e1d2a4acd9643d45ad80 Mon Sep 17 00:00:00 2001 From: nyov Date: Fri, 28 Feb 2020 02:10:13 +0000 Subject: [PATCH 1/3] Obsolete deprecated settings Obsolete REDIRECT_MAX_METAREFRESH_DELAY which has been deprecated since Scrapy 0.18 Obsolete LOG_UNSERIALIZABLE_REQUESTS which has been deprecated since Scrapy 1.2.0 and is replaced by SCHEDULER_DEBUG --- scrapy/core/scheduler.py | 3 +-- scrapy/downloadermiddlewares/redirect.py | 3 +-- scrapy/settings/deprecated.py | 2 -- tests/test_scheduler.py | 2 +- 4 files changed, 3 insertions(+), 7 deletions(-) diff --git a/scrapy/core/scheduler.py b/scrapy/core/scheduler.py index e184ed50e..c96b9b719 100644 --- a/scrapy/core/scheduler.py +++ b/scrapy/core/scheduler.py @@ -66,8 +66,7 @@ class Scheduler(object): dqclass = load_object(settings['SCHEDULER_DISK_QUEUE']) mqclass = load_object(settings['SCHEDULER_MEMORY_QUEUE']) - logunser = settings.getbool('LOG_UNSERIALIZABLE_REQUESTS', - settings.getbool('SCHEDULER_DEBUG')) + logunser = settings.getbool('SCHEDULER_DEBUG') return cls(dupefilter, jobdir=job_dir(settings), logunser=logunser, stats=crawler.stats, pqclass=pqclass, dqclass=dqclass, mqclass=mqclass, crawler=crawler) diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 77cb5aa94..08cff8a55 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -93,8 +93,7 @@ class MetaRefreshMiddleware(BaseRedirectMiddleware): def __init__(self, settings): super(MetaRefreshMiddleware, self).__init__(settings) self._ignore_tags = settings.getlist('METAREFRESH_IGNORE_TAGS') - self._maxdelay = settings.getint('REDIRECT_MAX_METAREFRESH_DELAY', - settings.getint('METAREFRESH_MAXDELAY')) + self._maxdelay = settings.getint('METAREFRESH_MAXDELAY') def process_response(self, request, response, spider): if request.meta.get('dont_redirect', False) or request.method == 'HEAD' or \ diff --git a/scrapy/settings/deprecated.py b/scrapy/settings/deprecated.py index 1211908df..f6f878725 100644 --- a/scrapy/settings/deprecated.py +++ b/scrapy/settings/deprecated.py @@ -11,8 +11,6 @@ DEPRECATED_SETTINGS = [ ('SQLITE_DB', 'no longer supported'), ('AUTOTHROTTLE_MIN_DOWNLOAD_DELAY', 'use DOWNLOAD_DELAY instead'), ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), - ('REDIRECT_MAX_METAREFRESH_DELAY', 'use METAREFRESH_MAXDELAY instead'), - ('LOG_UNSERIALIZABLE_REQUESTS', 'use SCHEDULER_DEBUG instead'), ] diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index e0e3600e5..13c297084 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -46,7 +46,7 @@ class MockCrawler(Crawler): def __init__(self, priority_queue_cls, jobdir): settings = dict( - LOG_UNSERIALIZABLE_REQUESTS=False, + SCHEDULER_DEBUG=False, SCHEDULER_DISK_QUEUE='scrapy.squeues.PickleLifoDiskQueue', SCHEDULER_MEMORY_QUEUE='scrapy.squeues.LifoMemoryQueue', SCHEDULER_PRIORITY_QUEUE=priority_queue_cls, From 64002255554aa2aa79863b657e5d1674d72b228d Mon Sep 17 00:00:00 2001 From: nyov Date: Fri, 28 Feb 2020 00:06:11 +0000 Subject: [PATCH 2/3] Drop horribly outdated deb package build files --- Makefile.buildbot | 24 -------------------- debian/changelog | 5 ----- debian/compat | 1 - debian/control | 20 ----------------- debian/copyright | 40 --------------------------------- debian/pyversions | 1 - debian/rules | 5 ----- debian/scrapy.docs | 2 -- debian/scrapy.install | 2 -- debian/scrapy.lintian-overrides | 1 - debian/scrapy.manpages | 1 - 11 files changed, 102 deletions(-) delete mode 100644 Makefile.buildbot delete mode 100644 debian/changelog delete mode 100644 debian/compat delete mode 100644 debian/control delete mode 100644 debian/copyright delete mode 100644 debian/pyversions delete mode 100755 debian/rules delete mode 100644 debian/scrapy.docs delete mode 100644 debian/scrapy.install delete mode 100644 debian/scrapy.lintian-overrides delete mode 100644 debian/scrapy.manpages diff --git a/Makefile.buildbot b/Makefile.buildbot deleted file mode 100644 index 775538259..000000000 --- a/Makefile.buildbot +++ /dev/null @@ -1,24 +0,0 @@ -TRIAL := $(shell which trial) -BRANCH := $(shell git rev-parse --abbrev-ref HEAD) -export PYTHONPATH=$(PWD) - -test: - coverage run --branch $(TRIAL) --reporter=text tests - rm -rf htmlcov && coverage html - -s3cmd sync -P htmlcov/ s3://static.scrapy.org/coverage-scrapy-$(BRANCH)/ - -build: - git describe --tags --match '[0-9]*' |sed 's/-/.post/;s/-g/+g/' >scrapy/VERSION - debchange -m -D unstable --force-distribution -v \ - $$(python setup.py --version |sed -r 's/([0-9]+.[0-9]+.[0-9]+)(a|b|rc|dev)([0-9]*)/\1~\2\3/')-$$(date +%s) \ - "Automatic build" - debuild -us -uc -b - -clean: - git checkout debian scrapy/VERSION - git clean -dfq - -pypi: - umask 0022 && chmod -R a+rX . && python setup.py sdist upload - -.PHONY: clean test build diff --git a/debian/changelog b/debian/changelog deleted file mode 100644 index dde97f9e3..000000000 --- a/debian/changelog +++ /dev/null @@ -1,5 +0,0 @@ -scrapy (0.11) unstable; urgency=low - - * Initial release. - - -- Scrapinghub Team Thu, 10 Jun 2010 17:24:02 -0300 diff --git a/debian/compat b/debian/compat deleted file mode 100644 index 7f8f011eb..000000000 --- a/debian/compat +++ /dev/null @@ -1 +0,0 @@ -7 diff --git a/debian/control b/debian/control deleted file mode 100644 index 2cc8eedf4..000000000 --- a/debian/control +++ /dev/null @@ -1,20 +0,0 @@ -Source: scrapy -Section: python -Priority: optional -Maintainer: Scrapinghub Team -Build-Depends: debhelper (>= 7.0.50), python (>=2.7), python-twisted, python-w3lib, python-lxml, python-six (>=1.5.2) -Standards-Version: 3.8.4 -Homepage: https://scrapy.org/ - -Package: scrapy -Architecture: all -Depends: ${python:Depends}, python-lxml, python-twisted, python-openssl, - python-w3lib (>= 1.8.0), python-queuelib, python-cssselect (>= 0.9), python-six (>=1.5.2) -Recommends: python-setuptools -Conflicts: python-scrapy, scrapy-0.25 -Provides: python-scrapy, scrapy-0.25 -Description: Python web crawling and web scraping framework - Scrapy is a fast high-level web crawling and web scraping framework, - used to crawl websites and extract structured data from their pages. - It can be used for a wide range of purposes, from data mining to - monitoring and automated testing. diff --git a/debian/copyright b/debian/copyright deleted file mode 100644 index c1bf47565..000000000 --- a/debian/copyright +++ /dev/null @@ -1,40 +0,0 @@ -This package was debianized by the Scrapinghub team . - -It was downloaded from https://scrapy.org - -Upstream Author: Scrapy Developers - -Copyright: 2007-2013 Scrapy Developers - -License: bsd - -Copyright (c) Scrapy developers. -All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, -are permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, - this list of conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright - notice, this list of conditions and the following disclaimer in the - documentation and/or other materials provided with the distribution. - - 3. Neither the name of Scrapy nor the names of its contributors may be used - to endorse or promote products derived from this software without - specific prior written permission. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND -ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED -WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE -DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR -ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES -(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; -LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT -(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS -SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The Debian packaging is (C) 2010-2013, Scrapinghub and -is licensed under the BSD, see `/usr/share/common-licenses/BSD'. diff --git a/debian/pyversions b/debian/pyversions deleted file mode 100644 index 1effb0034..000000000 --- a/debian/pyversions +++ /dev/null @@ -1 +0,0 @@ -2.7 diff --git a/debian/rules b/debian/rules deleted file mode 100755 index b8796e6e3..000000000 --- a/debian/rules +++ /dev/null @@ -1,5 +0,0 @@ -#!/usr/bin/make -f -# -*- makefile -*- - -%: - dh $@ diff --git a/debian/scrapy.docs b/debian/scrapy.docs deleted file mode 100644 index c19ffba4d..000000000 --- a/debian/scrapy.docs +++ /dev/null @@ -1,2 +0,0 @@ -README.rst -AUTHORS diff --git a/debian/scrapy.install b/debian/scrapy.install deleted file mode 100644 index c288ebed3..000000000 --- a/debian/scrapy.install +++ /dev/null @@ -1,2 +0,0 @@ -extras/scrapy_bash_completion etc/bash_completion.d/ -extras/scrapy_zsh_completion /usr/share/zsh/vendor-completions/_scrapy diff --git a/debian/scrapy.lintian-overrides b/debian/scrapy.lintian-overrides deleted file mode 100644 index b5de7f67d..000000000 --- a/debian/scrapy.lintian-overrides +++ /dev/null @@ -1 +0,0 @@ -new-package-should-close-itp-bug diff --git a/debian/scrapy.manpages b/debian/scrapy.manpages deleted file mode 100644 index 4818e9c92..000000000 --- a/debian/scrapy.manpages +++ /dev/null @@ -1 +0,0 @@ -extras/scrapy.1 From 6c35baae25517ed942c72d14d93363a389e5a9d3 Mon Sep 17 00:00:00 2001 From: nyov Date: Wed, 4 Mar 2020 00:40:11 +0000 Subject: [PATCH 3/3] Remove deprecated SiteNode and MultiValueDict classes --- scrapy/utils/datatypes.py | 174 -------------------------------------- 1 file changed, 174 deletions(-) diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index b07f995cf..175f92d77 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -6,183 +6,9 @@ This module must not depend on any module outside the Standard Library. """ import collections -import copy -import warnings import weakref from collections.abc import Mapping -from scrapy.exceptions import ScrapyDeprecationWarning - - -class MultiValueDictKeyError(KeyError): - def __init__(self, *args, **kwargs): - warnings.warn( - "scrapy.utils.datatypes.MultiValueDictKeyError is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2 - ) - super(MultiValueDictKeyError, self).__init__(*args, **kwargs) - - -class MultiValueDict(dict): - """ - A subclass of dictionary customized to handle multiple values for the same key. - - >>> d = MultiValueDict({'name': ['Adrian', 'Simon'], 'position': ['Developer']}) - >>> d['name'] - 'Simon' - >>> d.getlist('name') - ['Adrian', 'Simon'] - >>> d.get('lastname', 'nonexistent') - 'nonexistent' - >>> d.setlist('lastname', ['Holovaty', 'Willison']) - - This class exists to solve the irritating problem raised by cgi.parse_qs, - which returns a list for every key, even though most Web forms submit - single name-value pairs. - """ - def __init__(self, key_to_list_mapping=()): - warnings.warn("scrapy.utils.datatypes.MultiValueDict is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2) - dict.__init__(self, key_to_list_mapping) - - def __repr__(self): - return "<%s: %s>" % (self.__class__.__name__, dict.__repr__(self)) - - def __getitem__(self, key): - """ - Returns the last data value for this key, or [] if it's an empty list; - raises KeyError if not found. - """ - try: - list_ = dict.__getitem__(self, key) - except KeyError: - raise MultiValueDictKeyError("Key %r not found in %r" % (key, self)) - try: - return list_[-1] - except IndexError: - return [] - - def __setitem__(self, key, value): - dict.__setitem__(self, key, [value]) - - def __copy__(self): - return self.__class__(dict.items(self)) - - def __deepcopy__(self, memo=None): - if memo is None: - memo = {} - result = self.__class__() - memo[id(self)] = result - for key, value in dict.items(self): - dict.__setitem__(result, copy.deepcopy(key, memo), copy.deepcopy(value, memo)) - return result - - def get(self, key, default=None): - "Returns the default value if the requested data doesn't exist" - try: - val = self[key] - except KeyError: - return default - if val == []: - return default - return val - - def getlist(self, key): - "Returns an empty list if the requested data doesn't exist" - try: - return dict.__getitem__(self, key) - except KeyError: - return [] - - def setlist(self, key, list_): - dict.__setitem__(self, key, list_) - - def setdefault(self, key, default=None): - if key not in self: - self[key] = default - return self[key] - - def setlistdefault(self, key, default_list=()): - if key not in self: - self.setlist(key, default_list) - return self.getlist(key) - - def appendlist(self, key, value): - "Appends an item to the internal list associated with key" - self.setlistdefault(key, []) - dict.__setitem__(self, key, self.getlist(key) + [value]) - - def items(self): - """ - Returns a list of (key, value) pairs, where value is the last item in - the list associated with the key. - """ - return [(key, self[key]) for key in self.keys()] - - def lists(self): - "Returns a list of (key, list) pairs." - return dict.items(self) - - def values(self): - "Returns a list of the last value on every key list." - return [self[key] for key in self.keys()] - - def copy(self): - "Returns a copy of this object." - return self.__deepcopy__() - - def update(self, *args, **kwargs): - "update() extends rather than replaces existing key lists. Also accepts keyword args." - if len(args) > 1: - raise TypeError("update expected at most 1 arguments, got %d" % len(args)) - if args: - other_dict = args[0] - if isinstance(other_dict, MultiValueDict): - for key, value_list in other_dict.lists(): - self.setlistdefault(key, []).extend(value_list) - else: - try: - for key, value in other_dict.items(): - self.setlistdefault(key, []).append(value) - except TypeError: - raise ValueError("MultiValueDict.update() takes either a MultiValueDict or dictionary") - for key, value in kwargs.items(): - self.setlistdefault(key, []).append(value) - - -class SiteNode(object): - """Class to represent a site node (page, image or any other file)""" - - def __init__(self, url): - warnings.warn( - "scrapy.utils.datatypes.SiteNode is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2 - ) - - self.url = url - self.itemnames = [] - self.children = [] - self.parent = None - - def add_child(self, node): - self.children.append(node) - node.parent = self - - def to_string(self, level=0): - s = "%s%s\n" % (' ' * level, self.url) - if self.itemnames: - for n in self.itemnames: - s += "%sScraped: %s\n" % (' ' * (level + 1), n) - for node in self.children: - s += node.to_string(level + 1) - return s - class CaselessDict(dict):