From c025003da2c60a0a786c2ee158fe620402914273 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Sun, 18 Aug 2019 04:44:09 +0200 Subject: [PATCH 001/313] Add FTPFileStore --- scrapy/pipelines/files.py | 50 ++++++++++++++++++++++++++++++++++++--- 1 file changed, 47 insertions(+), 3 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 2145e6d2b..6f66460b8 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -10,6 +10,7 @@ import os.path import time import logging from email.utils import parsedate_tz, mktime_tz +from ftplib import FTP from six.moves.urllib.parse import urlparse from collections import defaultdict import six @@ -31,6 +32,7 @@ from scrapy.utils.python import to_bytes from scrapy.utils.request import referer_str from scrapy.utils.boto import is_botocore from scrapy.utils.datatypes import CaselessDict +from scrapy.utils.ftp import ftp_makedirs_cwd logger = logging.getLogger(__name__) @@ -248,6 +250,42 @@ class GCSFilesStore(object): ) +class FTPFilesStore(object): + + def __init__(self, uri): + assert uri.startswith('ftp://') + u = urlparse(uri) + self.ftp = FTP() + self.ftp.connect(u.hostname, u.port or '21') + self.ftp.login(u.username, u.password) + self.basedir = u.path + '/' + ftp_makedirs_cwd(self.ftp, self.basedir+'full') + + def persist_file(self, path, buf, info, meta=None, headers=None): + buf.seek(0) + filename = path.split('/')[1] + return threads.deferToThread( + self.ftp.storbinary, + 'STOR %s' % filename, + buf + ) + + def stat_file(self, path, info): + def _stat_file(path): + try: + last_modified = float(self.ftp.voidcmd("MDTM " + self.basedir + '/' + path)[4:].strip()) + m = hashlib.md5() + self.ftp.retrbinary('RETR %s' % self.basedir + path, m.update) + return {'last_modified': last_modified, 'checksum': m.hexdigest()} + # The file doesn't exist + except Exception as e : + return {} + return threads.deferToThread(_stat_file, path) + + def close_connection(self): + self.ftp.quit() + + class FilesPipeline(MediaPipeline): """Abstract pipeline that implement the file downloading @@ -274,6 +312,7 @@ class FilesPipeline(MediaPipeline): 'file': FSFilesStore, 's3': S3FilesStore, 'gs': GCSFilesStore, + 'ftp': FTPFilesStore } DEFAULT_FILES_URLS_FIELD = 'file_urls' DEFAULT_FILES_RESULT_FIELD = 'files' @@ -284,7 +323,6 @@ class FilesPipeline(MediaPipeline): if isinstance(settings, dict) or settings is None: settings = Settings(settings) - cls_name = "FilesPipeline" self.store = self._get_store(store_uri) resolve = functools.partial(self._key_for_pipe, @@ -303,7 +341,6 @@ class FilesPipeline(MediaPipeline): self.files_result_field = settings.get( resolve('FILES_RESULT_FIELD'), self.FILES_RESULT_FIELD ) - super(FilesPipeline, self).__init__(download_func=download_func, settings=settings) @classmethod @@ -320,7 +357,7 @@ class FilesPipeline(MediaPipeline): gcs_store = cls.STORE_SCHEMES['gs'] gcs_store.GCS_PROJECT_ID = settings['GCS_PROJECT_ID'] gcs_store.POLICY = settings['FILES_STORE_GCS_ACL'] or None - + store_uri = settings['FILES_STORE'] return cls(store_uri, settings=settings) @@ -461,3 +498,10 @@ class FilesPipeline(MediaPipeline): media_guid = hashlib.sha1(to_bytes(request.url)).hexdigest() media_ext = os.path.splitext(request.url)[1] return 'full/%s%s' % (media_guid, media_ext) + + def close_spider(self, spider): + try: + self.store.close_connection() + # If the store doesn't implement this function, pass + except AttributeError: + pass From 9b1587ed1bc736f9fcc357ce425405adf5bd6d08 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Mon, 19 Aug 2019 16:13:56 +0200 Subject: [PATCH 002/313] Credentials from settings-Support custom paths-Remove close conenction --- scrapy/pipelines/files.py | 35 ++++++++++++++++++++--------------- 1 file changed, 20 insertions(+), 15 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 6f66460b8..c22404799 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -251,19 +251,30 @@ class GCSFilesStore(object): class FTPFilesStore(object): + + FTP_USERNAME = None + FTP_PASSWORD = None def __init__(self, uri): assert uri.startswith('ftp://') u = urlparse(uri) self.ftp = FTP() self.ftp.connect(u.hostname, u.port or '21') - self.ftp.login(u.username, u.password) - self.basedir = u.path + '/' - ftp_makedirs_cwd(self.ftp, self.basedir+'full') + username = u.username or FTP_USERNAME + password = u.password or FTP_PASSWORD + self.ftp.login(username, password) + self.basedir = u.path def persist_file(self, path, buf, info, meta=None, headers=None): buf.seek(0) - filename = path.split('/')[1] + # If the path is like 'x/y/z.ext' the 'x/y' is rel_path and + # 'z.ext' is file name + # If path is only the file name 'z.ext', then rel_path is + # the empty string and filename is 'z.ext' + x = path.rsplit('/',1) + rel_path, filename = ('/' + x[0], x[1]) if len(x) > 1 else ('', x[0]) + abs_path = self.basedir + rel_path + ftp_makedirs_cwd(self.ftp, abs_path) return threads.deferToThread( self.ftp.storbinary, 'STOR %s' % filename, @@ -281,9 +292,6 @@ class FTPFilesStore(object): except Exception as e : return {} return threads.deferToThread(_stat_file, path) - - def close_connection(self): - self.ftp.quit() class FilesPipeline(MediaPipeline): @@ -357,6 +365,10 @@ class FilesPipeline(MediaPipeline): gcs_store = cls.STORE_SCHEMES['gs'] gcs_store.GCS_PROJECT_ID = settings['GCS_PROJECT_ID'] gcs_store.POLICY = settings['FILES_STORE_GCS_ACL'] or None + + ftp_store = cls.STORE_SCHEMES['ftp'] + ftp_store.FTP_USERNAME = settings['FTP_USER'] # Default is 'anonymous' + ftp_store.FTP_PASSWORD = settings['FTP_PASSWORD'] # Default is `guest` store_uri = settings['FILES_STORE'] return cls(store_uri, settings=settings) @@ -497,11 +509,4 @@ class FilesPipeline(MediaPipeline): def file_path(self, request, response=None, info=None): media_guid = hashlib.sha1(to_bytes(request.url)).hexdigest() media_ext = os.path.splitext(request.url)[1] - return 'full/%s%s' % (media_guid, media_ext) - - def close_spider(self, spider): - try: - self.store.close_connection() - # If the store doesn't implement this function, pass - except AttributeError: - pass + return 'full/%s%s' % (media_guid, media_ext) \ No newline at end of file From 0a5cb7745bc22b8e193c4b0e964b20e59ddc0bc2 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Mon, 19 Aug 2019 17:12:11 +0200 Subject: [PATCH 003/313] Fix reference mistake --- scrapy/pipelines/files.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index c22404799..cbe588f9a 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -260,8 +260,8 @@ class FTPFilesStore(object): u = urlparse(uri) self.ftp = FTP() self.ftp.connect(u.hostname, u.port or '21') - username = u.username or FTP_USERNAME - password = u.password or FTP_PASSWORD + username = u.username or self.FTP_USERNAME + password = u.password or self.FTP_PASSWORD self.ftp.login(username, password) self.basedir = u.path From 81ac1da3813c11a806aca845a5022e9964b03f80 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Mon, 19 Aug 2019 17:17:21 +0200 Subject: [PATCH 004/313] Handle leading and trailing slashes --- scrapy/pipelines/files.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index cbe588f9a..74697fc1d 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -263,7 +263,7 @@ class FTPFilesStore(object): username = u.username or self.FTP_USERNAME password = u.password or self.FTP_PASSWORD self.ftp.login(username, password) - self.basedir = u.path + self.basedir = u.path.rstrip('/') def persist_file(self, path, buf, info, meta=None, headers=None): buf.seek(0) @@ -272,7 +272,7 @@ class FTPFilesStore(object): # If path is only the file name 'z.ext', then rel_path is # the empty string and filename is 'z.ext' x = path.rsplit('/',1) - rel_path, filename = ('/' + x[0], x[1]) if len(x) > 1 else ('', x[0]) + rel_path, filename = ('/' + x[0].lstrip('/'), x[1]) if len(x) > 1 else ('', x[0]) abs_path = self.basedir + rel_path ftp_makedirs_cwd(self.ftp, abs_path) return threads.deferToThread( From 790bf9031229261639a5c457f6dd3498bdff8830 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Mon, 19 Aug 2019 19:16:47 +0200 Subject: [PATCH 005/313] Make FTP persiting files thread safe --- scrapy/pipelines/files.py | 43 +++++++++++++++++++++------------------ 1 file changed, 23 insertions(+), 20 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 74697fc1d..2959179b8 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -257,29 +257,32 @@ class FTPFilesStore(object): def __init__(self, uri): assert uri.startswith('ftp://') - u = urlparse(uri) - self.ftp = FTP() - self.ftp.connect(u.hostname, u.port or '21') - username = u.username or self.FTP_USERNAME - password = u.password or self.FTP_PASSWORD - self.ftp.login(username, password) + u = urlparse(uri) + self.port = u.port + self.host = u.hostname + self.port = int(u.port or '21') + self.username = u.username or self.FTP_USERNAME + self.password = u.password or self.FTP_PASSWORD self.basedir = u.path.rstrip('/') def persist_file(self, path, buf, info, meta=None, headers=None): - buf.seek(0) - # If the path is like 'x/y/z.ext' the 'x/y' is rel_path and - # 'z.ext' is file name - # If path is only the file name 'z.ext', then rel_path is - # the empty string and filename is 'z.ext' - x = path.rsplit('/',1) - rel_path, filename = ('/' + x[0].lstrip('/'), x[1]) if len(x) > 1 else ('', x[0]) - abs_path = self.basedir + rel_path - ftp_makedirs_cwd(self.ftp, abs_path) - return threads.deferToThread( - self.ftp.storbinary, - 'STOR %s' % filename, - buf - ) + + def _persist_file(path, buf): + ftp = FTP() + ftp.connect(self.host, self.port) + ftp.login(self.username, self.password) + buf.seek(0) + # If the path is like 'x/y/z.ext' the 'x/y' is rel_path and + # 'z.ext' is file name + # If path is only the file name 'z.ext', then rel_path is + # the empty string and filename is 'z.ext' + x = path.rsplit('/',1) + rel_path, filename = ('/' + x[0].lstrip('/'), x[1]) if len(x) > 1 else ('', x[0]) + abs_path = self.basedir + rel_path + ftp_makedirs_cwd(ftp, abs_path) + ftp.storbinary('STOR %s' % filename, buf) + + return threads.deferToThread(_persist_file, path, buf) def stat_file(self, path, info): def _stat_file(path): From 8c970c636eb37deeb5caddbe5069bfcc0a79015e Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Wed, 21 Aug 2019 18:28:36 +0200 Subject: [PATCH 006/313] port from str to int MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- scrapy/pipelines/files.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 2959179b8..bbe4b9558 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -260,7 +260,7 @@ class FTPFilesStore(object): u = urlparse(uri) self.port = u.port self.host = u.hostname - self.port = int(u.port or '21') + self.port = int(u.port or 21) self.username = u.username or self.FTP_USERNAME self.password = u.password or self.FTP_PASSWORD self.basedir = u.path.rstrip('/') @@ -512,4 +512,4 @@ class FilesPipeline(MediaPipeline): def file_path(self, request, response=None, info=None): media_guid = hashlib.sha1(to_bytes(request.url)).hexdigest() media_ext = os.path.splitext(request.url)[1] - return 'full/%s%s' % (media_guid, media_ext) \ No newline at end of file + return 'full/%s%s' % (media_guid, media_ext) From bd22b25ef4e4223aafc6e326058c6d62b1fbf13c Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Thu, 22 Aug 2019 01:30:15 +0200 Subject: [PATCH 007/313] Make `stat_file` thread safe .. Refactor file storing.. Support act/psv --- scrapy/extensions/feedexport.py | 17 +++++--------- scrapy/pipelines/files.py | 39 +++++++++++++++------------------ scrapy/utils/ftp.py | 21 +++++++++++++++++- 3 files changed, 44 insertions(+), 33 deletions(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index d35551fdd..1ddc55f93 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -19,7 +19,7 @@ from twisted.internet import defer, threads from w3lib.url import file_uri_to_path from scrapy import signals -from scrapy.utils.ftp import ftp_makedirs_cwd +from scrapy.utils.ftp import ftp_makedirs_cwd, ftp_store_file from scrapy.exceptions import NotConfigured from scrapy.utils.misc import create_instance, load_object from scrapy.utils.log import failure_to_exc_info @@ -174,16 +174,11 @@ class FTPFeedStorage(BlockingFeedStorage): ) def _store_in_thread(self, file): - file.seek(0) - ftp = FTP() - ftp.connect(self.host, self.port) - ftp.login(self.username, self.password) - if self.use_active_mode: - ftp.set_pasv(False) - dirname, filename = posixpath.split(self.path) - ftp_makedirs_cwd(ftp, dirname) - ftp.storbinary('STOR %s' % filename, file) - ftp.quit() + ftp_store_file( + self.path, file, self.host, + self.port, self.username, + self.password, self.use_active_mode + ) class SpiderSlot(object): diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index bbe4b9558..04fbf3237 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -32,7 +32,7 @@ from scrapy.utils.python import to_bytes from scrapy.utils.request import referer_str from scrapy.utils.boto import is_botocore from scrapy.utils.datatypes import CaselessDict -from scrapy.utils.ftp import ftp_makedirs_cwd +from scrapy.utils.ftp import ftp_makedirs_cwd, ftp_store_file logger = logging.getLogger(__name__) @@ -254,6 +254,7 @@ class FTPFilesStore(object): FTP_USERNAME = None FTP_PASSWORD = None + USE_ACTIVE_MODE = None def __init__(self, uri): assert uri.startswith('ftp://') @@ -265,31 +266,26 @@ class FTPFilesStore(object): self.password = u.password or self.FTP_PASSWORD self.basedir = u.path.rstrip('/') - def persist_file(self, path, buf, info, meta=None, headers=None): - - def _persist_file(path, buf): - ftp = FTP() - ftp.connect(self.host, self.port) - ftp.login(self.username, self.password) - buf.seek(0) - # If the path is like 'x/y/z.ext' the 'x/y' is rel_path and - # 'z.ext' is file name - # If path is only the file name 'z.ext', then rel_path is - # the empty string and filename is 'z.ext' - x = path.rsplit('/',1) - rel_path, filename = ('/' + x[0].lstrip('/'), x[1]) if len(x) > 1 else ('', x[0]) - abs_path = self.basedir + rel_path - ftp_makedirs_cwd(ftp, abs_path) - ftp.storbinary('STOR %s' % filename, buf) - - return threads.deferToThread(_persist_file, path, buf) + def persist_file(self, path, buf, info, meta=None, headers=None): + path = '%s/%s' % (self.basedir, path) + return threads.deferToThread( + ftp_store_file, path,buf, + self.host, self.port,self.username, + self.password, self.USE_ACTIVE_MODE + ) def stat_file(self, path, info): def _stat_file(path): try: - last_modified = float(self.ftp.voidcmd("MDTM " + self.basedir + '/' + path)[4:].strip()) + ftp = FTP() + ftp.connect(self.host, self.port) + ftp.login(self.username, self.password) + if self.USE_ACTIVE_MODE: + ftp.set_pasv(False) + file_path = "%s/%s" % (self.basedir, path) + last_modified = float(ftp.voidcmd("MDTM %s" % file_path)[4:].strip()) m = hashlib.md5() - self.ftp.retrbinary('RETR %s' % self.basedir + path, m.update) + ftp.retrbinary('RETR %s' % file_path, m.update) return {'last_modified': last_modified, 'checksum': m.hexdigest()} # The file doesn't exist except Exception as e : @@ -372,6 +368,7 @@ class FilesPipeline(MediaPipeline): ftp_store = cls.STORE_SCHEMES['ftp'] ftp_store.FTP_USERNAME = settings['FTP_USER'] # Default is 'anonymous' ftp_store.FTP_PASSWORD = settings['FTP_PASSWORD'] # Default is `guest` + ftp_store.USE_ACTIVE_MODE = settings.getbool('FEED_STORAGE_FTP_ACTIVE') store_uri = settings['FILES_STORE'] return cls(store_uri, settings=settings) diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index 9eca6a4da..ba94ec142 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -1,4 +1,6 @@ -from ftplib import error_perm +import posixpath + +from ftplib import error_perm, FTP from posixpath import dirname def ftp_makedirs_cwd(ftp, path, first_call=True): @@ -13,3 +15,20 @@ def ftp_makedirs_cwd(ftp, path, first_call=True): ftp.mkd(path) if first_call: ftp.cwd(path) + +def ftp_store_file( + path, file, host ,port, + username, password, use_active_mode=False): + """Opens a FTP connection with passed credentials,sets current directory + to the directory extracted from given path, then uploads the file to server + """ + ftp = FTP() + ftp.connect(host, port) + ftp.login(username, password) + if use_active_mode: + ftp.set_pasv(False) + file.seek(0) + dirname, filename = posixpath.split(path) + ftp_makedirs_cwd(ftp, dirname) + ftp.storbinary('STOR %s' % filename, file) + ftp.quit() \ No newline at end of file From 2047124b3573d02ec60b13614f4e3dce85e71546 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Thu, 22 Aug 2019 16:18:14 +0200 Subject: [PATCH 008/313] Follow PEP8 .. Remove unnecessary comments --- scrapy/pipelines/files.py | 6 ++++-- scrapy/utils/ftp.py | 2 +- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 04fbf3237..5780f63bd 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -330,6 +330,7 @@ class FilesPipeline(MediaPipeline): if isinstance(settings, dict) or settings is None: settings = Settings(settings) + cls_name = "FilesPipeline" self.store = self._get_store(store_uri) resolve = functools.partial(self._key_for_pipe, @@ -348,6 +349,7 @@ class FilesPipeline(MediaPipeline): self.files_result_field = settings.get( resolve('FILES_RESULT_FIELD'), self.FILES_RESULT_FIELD ) + super(FilesPipeline, self).__init__(download_func=download_func, settings=settings) @classmethod @@ -366,8 +368,8 @@ class FilesPipeline(MediaPipeline): gcs_store.POLICY = settings['FILES_STORE_GCS_ACL'] or None ftp_store = cls.STORE_SCHEMES['ftp'] - ftp_store.FTP_USERNAME = settings['FTP_USER'] # Default is 'anonymous' - ftp_store.FTP_PASSWORD = settings['FTP_PASSWORD'] # Default is `guest` + ftp_store.FTP_USERNAME = settings['FTP_USER'] + ftp_store.FTP_PASSWORD = settings['FTP_PASSWORD'] ftp_store.USE_ACTIVE_MODE = settings.getbool('FEED_STORAGE_FTP_ACTIVE') store_uri = settings['FILES_STORE'] diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index ba94ec142..bf67b9976 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -31,4 +31,4 @@ def ftp_store_file( dirname, filename = posixpath.split(path) ftp_makedirs_cwd(ftp, dirname) ftp.storbinary('STOR %s' % filename, file) - ftp.quit() \ No newline at end of file + ftp.quit() From 97d2f717ae30f53282a9cffacc40a879989f05af Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Thu, 22 Aug 2019 16:19:01 +0200 Subject: [PATCH 009/313] Support extracting ftp settings in `ImagesPipeline` --- scrapy/pipelines/images.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/scrapy/pipelines/images.py b/scrapy/pipelines/images.py index fa4d12ad1..872342fc0 100644 --- a/scrapy/pipelines/images.py +++ b/scrapy/pipelines/images.py @@ -99,6 +99,11 @@ class ImagesPipeline(FilesPipeline): gcs_store.GCS_PROJECT_ID = settings['GCS_PROJECT_ID'] gcs_store.POLICY = settings['IMAGES_STORE_GCS_ACL'] or None + ftp_store = cls.STORE_SCHEMES['ftp'] + ftp_store.FTP_USERNAME = settings['FTP_USER'] + ftp_store.FTP_PASSWORD = settings['FTP_PASSWORD'] + ftp_store.USE_ACTIVE_MODE = settings.getbool('FEED_STORAGE_FTP_ACTIVE') + store_uri = settings['IMAGES_STORE'] return cls(store_uri, settings=settings) From 0e8770a2f4e96ee18e2fd0fc7bfb5f9bcd2f623d Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Fri, 6 Sep 2019 15:47:57 +0200 Subject: [PATCH 010/313] test for files pipeline ftp store --- scrapy/utils/test.py | 18 ++++++++++++++++++ tests/test_pipeline_files.py | 25 ++++++++++++++++++++++++- 2 files changed, 42 insertions(+), 1 deletion(-) diff --git a/scrapy/utils/test.py b/scrapy/utils/test.py index 4b935c51b..59467f105 100644 --- a/scrapy/utils/test.py +++ b/scrapy/utils/test.py @@ -3,6 +3,7 @@ This module contains some assorted functions used in tests """ from __future__ import absolute_import +from posixpath import split import os from importlib import import_module @@ -61,6 +62,23 @@ def get_gcs_content_and_delete(bucket, path): bucket.delete_blob(path) return content, acl, blob +def get_ftp_content_and_delete(path, host ,port, + username, password, use_active_mode=False): + from ftplib import FTP + ftp = FTP() + ftp.connect(host, port) + ftp.login(username, password) + if use_active_mode: + ftp.set_pasv(False) + ftp_data = [] + def buffer_data(data): + ftp_data.append(data) + ftp.retrbinary('RETR %s' % path, buffer_data) + dirname, filename = split(path) + ftp.cwd(dirname) + ftp.delete(filename) + return "".join(ftp_data) + def get_crawler(spidercls=None, settings_dict=None): """Return an unconfigured Crawler object. If settings_dict is given, it will be used to populate the crawler settings with a project level diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index 0c5aaaa44..000c1e2e2 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -11,13 +11,14 @@ from six import BytesIO from twisted.trial import unittest from twisted.internet import defer -from scrapy.pipelines.files import FilesPipeline, FSFilesStore, S3FilesStore, GCSFilesStore +from scrapy.pipelines.files import FilesPipeline, FSFilesStore, S3FilesStore, GCSFilesStore, FTPFilesStore from scrapy.item import Item, Field from scrapy.http import Request, Response from scrapy.settings import Settings from scrapy.utils.python import to_bytes from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete from scrapy.utils.test import assert_gcs_environ, get_gcs_content_and_delete +from scrapy.utils.test import get_ftp_content_and_delete from scrapy.utils.boto import is_botocore from tests import mock @@ -365,6 +366,28 @@ class TestGCSFilesStore(unittest.TestCase): self.assertEqual(blob.content_type, 'application/octet-stream') self.assertIn(expected_policy, acl) +class TestFTPFileStore(unittest.TestCase): + @defer.inlineCallbacks + def test_persist(self): + uri = os.environ.get('FTP_TEST_FILE_URI') + if not uri: + raise unittest.SkipTest("No FTP URI available for testing") + data = b"TestFTPFilesStore: \xe2\x98\x83" + buf = BytesIO(data) + meta = {'foo': 'bar'} + path = 'full/filename' + store = FTPFilesStore(uri) + empty_dict = yield store.stat_file(path, info=None) + self.assertEqual(empty_dict, {}) + yield store.persist_file(path, buf, info=None, meta=meta, headers=None) + stat = yield store.stat_file(path, info=None) + self.assertIn('last_modified', stat) + self.assertIn('checksum', stat) + self.assertEqual(stat['checksum'], 'd113d66b2ec7258724a268bd88eef6b6') + path = '%s/%s' % (store.basedir, path) + content = get_ftp_content_and_delete(path, store.host, store.port, + store.username, store.password, store.USE_ACTIVE_MODE) + self.assertEqual(data.decode(), content) class ItemWithFiles(Item): file_urls = Field() From b14c3cb612becc28499409202e904c062745d52c Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Thu, 19 Sep 2019 23:33:57 +0200 Subject: [PATCH 011/313] Add media pipelines FTP documentation --- docs/topics/media-pipeline.rst | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 0ce431ff5..d3fed928c 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -148,6 +148,27 @@ Where: * ``full`` is a sub-directory to separate full images from thumbnails (if used). For more info see :ref:`topics-images-thumbnails`. +FTP server storage +------------------ + +.. setting:: FTP_USER +.. setting:: FTP_PASSWORD + +:setting:`FILES_STORE` and :setting:`IMAGES_STORE` can represent a FTP server. +Scrapy will automatically upload the files to the server. + +:setting:`FILES_STORE` value: should be written in the form +`ftp://username:password@address:port/path` or `ftp://address:port/path`. In +the second case, the `username` and `password` are taken from `FTP_USER` and +`FTP_PASSWORD` settings respectively. + +.. note:: + The `path` can be left empty + +FTP supports two different connection modes: active or passive. Scrapy uses +the passive connection mode by default. To use the active connection mode instead, +set the `FEED_STORAGE_FTP_ACTIVE` setting to True. + Amazon S3 storage ----------------- From 28005b2872b897d84343d1e145fe50be880e91ff Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Sat, 28 Sep 2019 06:21:14 +0200 Subject: [PATCH 012/313] Update media-pipeline.rst --- docs/topics/media-pipeline.rst | 20 +++++++++----------- 1 file changed, 9 insertions(+), 11 deletions(-) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index d3fed928c..ceac317c0 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -151,23 +151,21 @@ Where: FTP server storage ------------------ -.. setting:: FTP_USER -.. setting:: FTP_PASSWORD - -:setting:`FILES_STORE` and :setting:`IMAGES_STORE` can represent a FTP server. +:setting:`FILES_STORE` and :setting:`IMAGES_STORE` can point to an FTP server. Scrapy will automatically upload the files to the server. -:setting:`FILES_STORE` value: should be written in the form -`ftp://username:password@address:port/path` or `ftp://address:port/path`. In -the second case, the `username` and `password` are taken from `FTP_USER` and -`FTP_PASSWORD` settings respectively. +:setting:`FILES_STORE` and :setting:`IMAGES_STORE` should be written in one of the +following forms:: -.. note:: - The `path` can be left empty + ftp://username:password@address:port/path + ftp://address:port/path + +If ``username`` and ``password`` are not provided, they are taken from :setting:`FTP_USER` and +:setting:`FTP_PASSWORD` settings respectively. FTP supports two different connection modes: active or passive. Scrapy uses the passive connection mode by default. To use the active connection mode instead, -set the `FEED_STORAGE_FTP_ACTIVE` setting to True. +set the :setting:`FEED_STORAGE_FTP_ACTIVE` setting to ``True``. Amazon S3 storage ----------------- From 7f4f98fd38d3fdf6b45a9d0289df0cbb48bcd22b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 30 Sep 2019 18:22:28 +0200 Subject: [PATCH 013/313] Provide complete API documentation coverage of scrapy.linkextractors --- docs/conf.py | 4 +++ docs/topics/link-extractors.rst | 45 ++++++++++++------------------- scrapy/linkextractors/__init__.py | 11 ++++++++ scrapy/linkextractors/lxmlhtml.py | 8 ++++++ tests/test_linkextractors.py | 30 +++++++++++++++++++++ 5 files changed, 70 insertions(+), 28 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 34dd5bcb7..5ba6b8d43 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -265,6 +265,10 @@ coverage_ignore_pyobjects = [ # Never documented before, and deprecated now. r'^scrapy\.item\.DictItem$', + r'^scrapy\.linkextractors\.FilteringLinkExtractor$', + + # Implementation detail of LxmlLinkExtractor + r'^scrapy\.linkextractors\.lxmlhtml\.LxmlParserLinkExtractor', ] diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index 713a94e10..f9936a498 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -4,46 +4,33 @@ Link Extractors =============== -Link extractors are objects whose only purpose is to extract links from web -pages (:class:`scrapy.http.Response` objects) which will be eventually -followed. +A link extractor is an object that extracts links from responses. -There is ``scrapy.linkextractors.LinkExtractor`` available -in Scrapy, but you can create your own custom Link Extractors to suit your -needs by implementing a simple interface. - -The only public method that every link extractor has is ``extract_links``, -which receives a :class:`~scrapy.http.Response` object and returns a list -of :class:`scrapy.link.Link` objects. Link extractors are meant to be -instantiated once and their ``extract_links`` method called several times -with different responses to extract links to follow. - -Link extractors are used in the :class:`~scrapy.spiders.CrawlSpider` -class (available in Scrapy), through a set of rules, but you can also use it in -your spiders, even if you don't subclass from -:class:`~scrapy.spiders.CrawlSpider`, as its purpose is very simple: to -extract links. +The constructor of :class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` +takes settings that determine which links may be extracted. +:class:`LxmlLinkExtractor.extract_links +` returns a +list of matching :class:`scrapy.link.Link` objects from a +:class:`~scrapy.http.Response` object. +Link extractors are used in :class:`~scrapy.spiders.CrawlSpider` spiders +through a set of :class:`~scrapy.spiders.Rule` objects. You can also use link +extractors in regular spiders. .. _topics-link-extractors-ref: -Built-in link extractors reference -================================== +Link extractor reference +======================== .. module:: scrapy.linkextractors :synopsis: Link extractors classes -Link extractors classes bundled with Scrapy are provided in the -:mod:`scrapy.linkextractors` module. - -The default link extractor is ``LinkExtractor``, which is the same as -:class:`~.LxmlLinkExtractor`:: +The link extractor class is +:class:`scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`. For convenience it +can also be imported as ``scrapy.linkextractors.LinkExtractor``:: from scrapy.linkextractors import LinkExtractor -There used to be other link extractor classes in previous Scrapy versions, -but they are deprecated now. - LxmlLinkExtractor ----------------- @@ -152,4 +139,6 @@ LxmlLinkExtractor from elements or attributes which allow leading/trailing whitespaces). :type strip: boolean + .. automethod:: extract_links + .. _scrapy.linkextractors: https://github.com/scrapy/scrapy/blob/master/scrapy/linkextractors/__init__.py diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index ebf3cd7d8..ca80dc339 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -6,11 +6,13 @@ This package contains a collection of Link Extractors. For more info see docs/topics/link-extractors.rst """ import re +from warnings import warn from six.moves.urllib.parse import urlparse from parsel.csstranslator import HTMLTranslator from w3lib.url import canonicalize_url +from scrapy.utils.deprecate import ScrapyDeprecationWarning from scrapy.utils.misc import arg_to_iter from scrapy.utils.url import ( url_is_from_any_domain, url_has_any_extension, @@ -49,6 +51,15 @@ class FilteringLinkExtractor(object): _csstranslator = HTMLTranslator() + def __new__(cls, *args, **kwargs): + from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor + if (issubclass(cls, FilteringLinkExtractor) and + not issubclass(cls, LxmlLinkExtractor)): + warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' + 'please use scrapy.linkextractors.LinkExtractor instead', + ScrapyDeprecationWarning, stacklevel=2) + return super(FilteringLinkExtractor, cls).__new__(cls) + def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text): diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index 8f6f93a44..41091ba23 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -117,6 +117,14 @@ class LxmlLinkExtractor(FilteringLinkExtractor): restrict_text=restrict_text) def extract_links(self, response): + """Returns a list of :class:`~scrapy.link.Link` objects from the + specified :class:`response `. + + Only links that match the settings passed to the link extractor + constructor are returned. + + Duplicate links are omitted. + """ base_url = get_base_url(response) if self.restrict_xpaths: docs = [subdoc diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index d96e259f6..ea6db28c0 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -1,10 +1,13 @@ import re import unittest +from warnings import catch_warnings import pytest +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import HtmlResponse, XmlResponse from scrapy.link import Link +from scrapy.linkextractors import FilteringLinkExtractor from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor from tests import get_testdata @@ -506,3 +509,30 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): @pytest.mark.xfail def test_restrict_xpaths_with_html_entities(self): super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities() + + def test_filteringlinkextractor_deprecation_warning(self): + """Make sure the FilteringLinkExtractor deprecation warning is not + issued for LxmlLinkExtractor""" + with catch_warnings(record=True) as warnings: + extractor = LxmlLinkExtractor() + self.assertEqual(len(warnings), 0) + class SubclassedItem(LxmlLinkExtractor): + pass + subclassed_extractor = SubclassedItem() + self.assertEqual(len(warnings), 0) + + +class FilteringLinkExtractorTest(unittest.TestCase): + + def test_deprecation_warning(self): + args = [None]*10 + with catch_warnings(record=True) as warnings: + extractor = FilteringLinkExtractor(*args) + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) + with catch_warnings(record=True) as warnings: + class SubclassedFilteringLinkExtractor(FilteringLinkExtractor): + pass + subclassed_extractor = SubclassedFilteringLinkExtractor(*args) + self.assertEqual(len(warnings), 1) + self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) From 175cd2ece5b9a8e2c735697e7a49d0baffc7cd52 Mon Sep 17 00:00:00 2001 From: OmarFarrag Date: Tue, 1 Oct 2019 07:27:31 +0200 Subject: [PATCH 014/313] Update docs/topics/media-pipeline.rst MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- docs/topics/media-pipeline.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index ceac317c0..327517996 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -160,7 +160,7 @@ following forms:: ftp://username:password@address:port/path ftp://address:port/path -If ``username`` and ``password`` are not provided, they are taken from :setting:`FTP_USER` and +If ``username`` and ``password`` are not provided, they are taken from the :setting:`FTP_USER` and :setting:`FTP_PASSWORD` settings respectively. FTP supports two different connection modes: active or passive. Scrapy uses From 5f168cd459cfc026de1e4d0b43c45ea740fe1dd5 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 2 Oct 2019 14:08:08 -0300 Subject: [PATCH 015/313] Response.follow_all --- docs/topics/request-response.rst | 4 ++ scrapy/http/response/__init__.py | 63 +++++++++++++++++----- scrapy/http/response/text.py | 61 ++++++++++++++++++--- tests/test_http_response.py | 91 ++++++++++++++++++++++++++++++++ 4 files changed, 199 insertions(+), 20 deletions(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 727c67482..9fe3c7518 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -701,6 +701,8 @@ Response objects .. automethod:: Response.follow + .. automethod:: Response.follow_all + .. _urlparse.urljoin: https://docs.python.org/2/library/urlparse.html#urlparse.urljoin @@ -790,6 +792,8 @@ TextResponse objects .. automethod:: TextResponse.follow + .. automethod:: TextResponse.follow_all + .. method:: TextResponse.body_as_unicode() The same as :attr:`text`, but available as a method. This method is diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index b0a526b72..96359705e 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -113,8 +113,8 @@ class Response(object_ref): It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be a relative URL or a ``scrapy.link.Link`` object, not only an absolute URL. - - :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` + + :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` method which supports selectors in addition to absolute/relative URLs and Link objects. """ @@ -123,14 +123,51 @@ class Response(object_ref): elif url is None: raise ValueError("url can't be None") url = self.urljoin(url) - return Request(url, callback, - method=method, - headers=headers, - body=body, - cookies=cookies, - meta=meta, - encoding=encoding, - priority=priority, - dont_filter=dont_filter, - errback=errback, - cb_kwargs=cb_kwargs) + return Request( + url=url, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, + ) + + def follow_all(self, urls, callback=None, method='GET', headers=None, body=None, + cookies=None, meta=None, encoding='utf-8', priority=0, + dont_filter=False, errback=None, cb_kwargs=None): + # type: (...) -> Generator[Request, None, None] + """ + Return an iterable of :class:`~.Request` instance to follow all links + in ``urls``. It accepts the same arguments as ``Request.__init__`` method, + but elements of ``urls`` can be relative URLs or ``scrapy.link.Link`` objects + not only absolute URLs. + + :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow_all` + method which supports selectors in addition to absolute/relative URLs + and Link objects. + """ + if not hasattr(urls, '__iter__'): + raise TypeError("'urls' argument must be an iterable") + return ( + self.follow( + url=url, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, + ) + for url in urls + ) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 339913d4e..81e3f9b28 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -13,7 +13,6 @@ from w3lib.encoding import html_to_unicode, resolve_encoding, \ html_body_declared_encoding, http_content_type_encoding from w3lib.html import strip_html5_whitespace -from scrapy.http.request import Request from scrapy.http.response import Response from scrapy.utils.response import get_base_url from scrapy.utils.python import memoizemethod_noargs, to_native_str @@ -44,7 +43,7 @@ class TextResponse(Response): if isinstance(body, six.text_type): if self._encoding is None: raise TypeError('Cannot convert unicode body - %s has no encoding' % - type(self).__name__) + type(self).__name__) self._body = body.encode(self._encoding) else: super(TextResponse, self)._set_body(body) @@ -90,8 +89,8 @@ class TextResponse(Response): if self._cached_benc is None: content_type = to_native_str(self.headers.get(b'Content-Type', b'')) benc, ubody = html_to_unicode(content_type, self.body, - auto_detect_fun=self._auto_detect_fun, - default_encoding=self._DEFAULT_ENCODING) + auto_detect_fun=self._auto_detect_fun, + default_encoding=self._DEFAULT_ENCODING) self._cached_benc = benc self._cached_ubody = ubody return self._cached_benc @@ -129,7 +128,7 @@ class TextResponse(Response): Return a :class:`~.Request` instance to follow a link ``url``. It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be not only an absolute URL, but also - + * a relative URL; * a scrapy.link.Link object (e.g. a link extractor result); * an attribute Selector (not SelectorList) - e.g. @@ -137,7 +136,7 @@ class TextResponse(Response): ``response.xpath('//img/@src')[0]``. * a Selector for ```` or ```` element, e.g. ``response.css('a.my_link')[0]``. - + See :ref:`response-follow-example` for usage examples. """ if isinstance(url, parsel.Selector): @@ -145,7 +144,9 @@ class TextResponse(Response): elif isinstance(url, parsel.SelectorList): raise ValueError("SelectorList is not supported") encoding = self.encoding if encoding is None else encoding - return super(TextResponse, self).follow(url, callback, + return super(TextResponse, self).follow( + url=url, + callback=callback, method=method, headers=headers, body=body, @@ -158,6 +159,52 @@ class TextResponse(Response): cb_kwargs=cb_kwargs, ) + def follow_all(self, urls=None, callback=None, method='GET', headers=None, body=None, + cookies=None, meta=None, encoding=None, priority=0, + dont_filter=False, errback=None, cb_kwargs=None, + css=None, xpath=None): + # type: (...) -> Generator[Request, None, None] + """ + A generator that produces :class:`~.Request` instances to follow all + links in ``urls``. It accepts the same arguments as the :class:`~.Request` + initializer, except that each ``urls`` element does not need to be an absolute + URL, it can be any of the following: + + * a relative URL; + * a scrapy.link.Link object (e.g. a link extractor result); + * an attribute Selector (not SelectorList) - e.g. + ``response.css('a::attr(href)')[0]`` or + ``response.xpath('//img/@src')[0]``. + * a Selector for ```` or ```` element, e.g. + ``response.css('a.my_link')[0]``. + + In addition, ``css`` and ``xpath`` arguments are accepted to perform the link extraction + within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` are accepted). + """ + if len(list(filter(None, (urls, css, xpath)))) > 1: + raise ValueError('Please supply only one of the following arguments: {urls, css, xpath}') + if css: + urls = self.css(css) + elif xpath: + urls = self.xpath(xpath) + return ( + self.follow( + url=url, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, + ) + for url in urls + ) + def _url_from_selector(sel): # type: (parsel.Selector) -> str diff --git a/tests/test_http_response.py b/tests/test_http_response.py index cd5c3486e..134856c78 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -143,6 +143,8 @@ class BaseResponseTest(unittest.TestCase): r.css('body') r.xpath('//body') + # Response.follow + def test_follow_url_absolute(self): self._assert_followed_url('http://foo.example.com', 'http://foo.example.com') @@ -166,6 +168,64 @@ class BaseResponseTest(unittest.TestCase): def test_follow_whitespace_link(self): self._assert_followed_url(Link('http://example.com/foo '), 'http://example.com/foo%20') + + # Response.follow_all + + def test_follow_all_absolute(self): + url_list = ['http://example.org', 'http://www.example.org', + 'http://example.com', 'http://www.example.com'] + self._assert_followed_all_urls(url_list, url_list) + + def test_follow_all_relative(self): + relative = ['foo', 'bar', 'foo/bar', 'bar/foo'] + absolute = [ + 'http://example.com/foo', + 'http://example.com/bar', + 'http://example.com/foo/bar', + 'http://example.com/bar/foo', + ] + self._assert_followed_all_urls(relative, absolute) + + def test_follow_all_links(self): + absolute = [ + 'http://example.com/foo', + 'http://example.com/bar', + 'http://example.com/foo/bar', + 'http://example.com/bar/foo', + ] + links = map(Link, absolute) + self._assert_followed_all_urls(links, absolute) + + def test_follow_all_invalid(self): + r = self.response_class("http://example.com") + with self.assertRaises(TypeError): + list(r.follow_all(None)) + with self.assertRaises(TypeError): + list(r.follow_all(12345)) + with self.assertRaises(ValueError): + list(r.follow_all([None])) + + def test_follow_all_whitespace(self): + relative = ['foo ', 'bar ', 'foo/bar ', 'bar/foo '] + absolute = [ + 'http://example.com/foo%20', + 'http://example.com/bar%20', + 'http://example.com/foo/bar%20', + 'http://example.com/bar/foo%20', + ] + self._assert_followed_all_urls(relative, absolute) + + def test_follow_all_whitespace_links(self): + absolute = [ + 'http://example.com/foo ', + 'http://example.com/bar ', + 'http://example.com/foo/bar ', + 'http://example.com/bar/foo ', + ] + links = map(Link, absolute) + expected = [u.replace(' ', '%20') for u in absolute] + self._assert_followed_all_urls(links, expected) + def _assert_followed_url(self, follow_obj, target_url, response=None): if response is None: response = self._links_response() @@ -173,6 +233,14 @@ class BaseResponseTest(unittest.TestCase): self.assertEqual(req.url, target_url) return req + def _assert_followed_all_urls(self, follow_obj, target_urls, response=None): + if response is None: + response = self._links_response() + followed = response.follow_all(follow_obj) + for req, target in zip(followed, target_urls): + self.assertEqual(req.url, target) + yield req + def _links_response(self): body = get_testdata('link_extractor', 'sgml_linkextractor.html') resp = self.response_class('http://example.com/index', body=body) @@ -483,6 +551,29 @@ class TextResponseTest(BaseResponseTest): ) self.assertEqual(req.encoding, 'cp1251') + def test_follow_all_css(self): + expected = [ + 'http://example.com/sample3.html', + 'http://example.com/innertag.html', + ] + response = self._links_response() + extracted = [r.url for r in response.follow_all(css='a[href*="example.com"]')] + self.assertEqual(expected, extracted) + + def test_follow_all_xpath(self): + expected = [ + 'http://example.com/sample3.html', + 'http://example.com/innertag.html', + ] + response = self._links_response() + extracted = response.follow_all(xpath='//a[contains(@href, "example.com")]') + self.assertEqual(expected, [r.url for r in extracted]) + + def test_follow_all_exception(self): + response = self._links_response() + with self.assertRaises(ValueError): + response.follow_all(css='a[href*="example.com"]', xpath='//a[contains(@href, "example.com")]') + class HtmlResponseTest(TextResponseTest): From e1fa1fd8ad62cb8958d21d3a33648ed4de6af757 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Thu, 10 Oct 2019 00:36:38 -0300 Subject: [PATCH 016/313] TextResponse.follow_all: skip invalid links --- scrapy/http/response/text.py | 26 +++++++--- .../sgml_linkextractor_no_href.html | 25 ++++++++++ tests/test_http_response.py | 47 ++++++++++++++++--- 3 files changed, 85 insertions(+), 13 deletions(-) create mode 100644 tests/sample_data/link_extractor/sgml_linkextractor_no_href.html diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 81e3f9b28..ccacce550 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -180,13 +180,27 @@ class TextResponse(Response): In addition, ``css`` and ``xpath`` arguments are accepted to perform the link extraction within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` are accepted). + + Note that when using the ``css`` or ``xpath`` parameters, this method will not produce + requests for selectors from which links cannot be obtained (for instance, anchor tags + without ``href`` attribute) """ - if len(list(filter(None, (urls, css, xpath)))) > 1: - raise ValueError('Please supply only one of the following arguments: {urls, css, xpath}') - if css: - urls = self.css(css) - elif xpath: - urls = self.xpath(xpath) + arg_count = len(list(filter(None, (urls, css, xpath)))) + if arg_count != 1: + raise ValueError('Please supply exactly one of the following arguments: {urls, css, xpath}') + if not urls: + urls = [] + if css: + selector_method = getattr(self, 'css') + expression = css + elif xpath: + selector_method = getattr(self, 'xpath') + expression = xpath + for selector in selector_method(expression): + try: + urls.append(_url_from_selector(selector)) + except ValueError: + pass return ( self.follow( url=url, diff --git a/tests/sample_data/link_extractor/sgml_linkextractor_no_href.html b/tests/sample_data/link_extractor/sgml_linkextractor_no_href.html new file mode 100644 index 000000000..0b01cede8 --- /dev/null +++ b/tests/sample_data/link_extractor/sgml_linkextractor_no_href.html @@ -0,0 +1,25 @@ + + + + Sample page with anchor tags containing no href attribute, to test the TextResponse.follow_all method + + + + + + + \ No newline at end of file diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 134856c78..6d3c5cb9d 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -198,12 +198,20 @@ class BaseResponseTest(unittest.TestCase): def test_follow_all_invalid(self): r = self.response_class("http://example.com") - with self.assertRaises(TypeError): - list(r.follow_all(None)) - with self.assertRaises(TypeError): - list(r.follow_all(12345)) - with self.assertRaises(ValueError): - list(r.follow_all([None])) + if self.response_class == Response: + with self.assertRaises(TypeError): + list(r.follow_all(urls=None)) + with self.assertRaises(TypeError): + list(r.follow_all(urls=12345)) + with self.assertRaises(ValueError): + list(r.follow_all(urls=[None])) + else: + with self.assertRaises(ValueError): + list(r.follow_all(urls=None)) + with self.assertRaises(TypeError): + list(r.follow_all(urls=12345)) + with self.assertRaises(ValueError): + list(r.follow_all(urls=[None])) def test_follow_all_whitespace(self): relative = ['foo ', 'bar ', 'foo/bar ', 'bar/foo '] @@ -246,6 +254,11 @@ class BaseResponseTest(unittest.TestCase): resp = self.response_class('http://example.com/index', body=body) return resp + def _links_response_no_href(self): + body = get_testdata('link_extractor', 'sgml_linkextractor_no_href.html') + resp = self.response_class('http://example.com/index', body=body) + return resp + class TextResponseTest(BaseResponseTest): @@ -560,6 +573,16 @@ class TextResponseTest(BaseResponseTest): extracted = [r.url for r in response.follow_all(css='a[href*="example.com"]')] self.assertEqual(expected, extracted) + def test_follow_all_css_skip_invalid(self): + expected = [ + 'http://example.com/page/1/', + 'http://example.com/page/3/', + 'http://example.com/page/4/', + ] + response = self._links_response_no_href() + extracted = [r.url for r in response.follow_all(css='.pagination a')] + self.assertEqual(expected, extracted) + def test_follow_all_xpath(self): expected = [ 'http://example.com/sample3.html', @@ -569,7 +592,17 @@ class TextResponseTest(BaseResponseTest): extracted = response.follow_all(xpath='//a[contains(@href, "example.com")]') self.assertEqual(expected, [r.url for r in extracted]) - def test_follow_all_exception(self): + def test_follow_all_xpath_skip_invalid(self): + expected = [ + 'http://example.com/page/1/', + 'http://example.com/page/3/', + 'http://example.com/page/4/', + ] + response = self._links_response_no_href() + extracted = [r.url for r in response.follow_all(xpath='//div[@id="pagination"]/a')] + self.assertEqual(expected, extracted) + + def test_follow_all_too_many_arguments(self): response = self._links_response() with self.assertRaises(ValueError): response.follow_all(css='a[href*="example.com"]', xpath='//a[contains(@href, "example.com")]') From b970851299bf561af7ff0bb21caca9b6c3f296af Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 14 Oct 2019 13:35:06 -0300 Subject: [PATCH 017/313] Update scrapy/http/response/__init__.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- scrapy/http/response/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index 96359705e..cdaababac 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -143,7 +143,7 @@ class Response(object_ref): dont_filter=False, errback=None, cb_kwargs=None): # type: (...) -> Generator[Request, None, None] """ - Return an iterable of :class:`~.Request` instance to follow all links + Return an iterable of :class:`~.Request` instances to follow all links in ``urls``. It accepts the same arguments as ``Request.__init__`` method, but elements of ``urls`` can be relative URLs or ``scrapy.link.Link`` objects not only absolute URLs. From ba840c5a6bea95333b3b78fb767ac6f092d02177 Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 14 Oct 2019 13:35:24 -0300 Subject: [PATCH 018/313] Update scrapy/http/response/__init__.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- scrapy/http/response/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index cdaababac..92fa01621 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -145,7 +145,7 @@ class Response(object_ref): """ Return an iterable of :class:`~.Request` instances to follow all links in ``urls``. It accepts the same arguments as ``Request.__init__`` method, - but elements of ``urls`` can be relative URLs or ``scrapy.link.Link`` objects + but elements of ``urls`` can be relative URLs or :class:`~scrapy.link.Link` objects, not only absolute URLs. :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow_all` From 498d33aac37d5d7a0955e9b409fc3d5ff8627dc0 Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 14 Oct 2019 13:35:54 -0300 Subject: [PATCH 019/313] Update scrapy/http/response/text.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- scrapy/http/response/text.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index ccacce550..bb5a4eb9d 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -183,7 +183,7 @@ class TextResponse(Response): Note that when using the ``css`` or ``xpath`` parameters, this method will not produce requests for selectors from which links cannot be obtained (for instance, anchor tags - without ``href`` attribute) + without an ``href`` attribute) """ arg_count = len(list(filter(None, (urls, css, xpath)))) if arg_count != 1: From c7c54f5453eaa16f2b6d6441be68b0a272606ec9 Mon Sep 17 00:00:00 2001 From: elacuesta Date: Mon, 14 Oct 2019 13:47:44 -0300 Subject: [PATCH 020/313] Update scrapy/http/response/text.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- scrapy/http/response/text.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index bb5a4eb9d..ecb582b4a 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -179,7 +179,7 @@ class TextResponse(Response): ``response.css('a.my_link')[0]``. In addition, ``css`` and ``xpath`` arguments are accepted to perform the link extraction - within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` are accepted). + within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` is accepted). Note that when using the ``css`` or ``xpath`` parameters, this method will not produce requests for selectors from which links cannot be obtained (for instance, anchor tags From 9d5398e7f2834940cbf9c2efa312bf5d96ffba98 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Mon, 14 Oct 2019 13:57:46 -0300 Subject: [PATCH 021/313] TextResponse.follow_all: improve docs --- scrapy/http/response/text.py | 28 +++++++++++++++------------- 1 file changed, 15 insertions(+), 13 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index ecb582b4a..b2907baa4 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -129,13 +129,14 @@ class TextResponse(Response): It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be not only an absolute URL, but also - * a relative URL; - * a scrapy.link.Link object (e.g. a link extractor result); - * an attribute Selector (not SelectorList) - e.g. + * a relative URL + * a :class:`~scrapy.link.Link` object, e.g. the result of + :ref:`topics-link-extractors` + * a :class:`~scrapy.selector.Selector` object for a ```` or ```` element, e.g. + ``response.css('a.my_link')[0]`` + * an attribute :class:`~scrapy.selector.Selector` (not SelectorList), e.g. ``response.css('a::attr(href)')[0]`` or - ``response.xpath('//img/@src')[0]``. - * a Selector for ```` or ```` element, e.g. - ``response.css('a.my_link')[0]``. + ``response.xpath('//img/@src')[0]`` See :ref:`response-follow-example` for usage examples. """ @@ -170,13 +171,14 @@ class TextResponse(Response): initializer, except that each ``urls`` element does not need to be an absolute URL, it can be any of the following: - * a relative URL; - * a scrapy.link.Link object (e.g. a link extractor result); - * an attribute Selector (not SelectorList) - e.g. + * a relative URL + * a :class:`~scrapy.link.Link` object, e.g. the result of + :ref:`topics-link-extractors` + * a :class:`~scrapy.selector.Selector` object for a ```` or ```` element, e.g. + ``response.css('a.my_link')[0]`` + * an attribute :class:`~scrapy.selector.Selector` (not SelectorList), e.g. ``response.css('a::attr(href)')[0]`` or - ``response.xpath('//img/@src')[0]``. - * a Selector for ```` or ```` element, e.g. - ``response.css('a.my_link')[0]``. + ``response.xpath('//img/@src')[0]`` In addition, ``css`` and ``xpath`` arguments are accepted to perform the link extraction within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` is accepted). @@ -187,7 +189,7 @@ class TextResponse(Response): """ arg_count = len(list(filter(None, (urls, css, xpath)))) if arg_count != 1: - raise ValueError('Please supply exactly one of the following arguments: {urls, css, xpath}') + raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath') if not urls: urls = [] if css: From 2a4d4a466aa6cede4829b62c7287332a627877ed Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 15 Oct 2019 11:52:12 -0300 Subject: [PATCH 022/313] TextResponse.follow_all: Simplify implementation --- scrapy/http/response/text.py | 12 +++++------- 1 file changed, 5 insertions(+), 7 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index b2907baa4..25e115bf9 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -191,14 +191,12 @@ class TextResponse(Response): if arg_count != 1: raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath') if not urls: - urls = [] if css: - selector_method = getattr(self, 'css') - expression = css - elif xpath: - selector_method = getattr(self, 'xpath') - expression = xpath - for selector in selector_method(expression): + selector_list = self.css(css) + if xpath: + selector_list = self.xpath(xpath) + urls = [] + for selector in selector_list: try: urls.append(_url_from_selector(selector)) except ValueError: From 2c6f7fee6456168f4870293c442258a2470b1b48 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 15 Oct 2019 13:48:14 -0300 Subject: [PATCH 023/313] TextResponse.follow_all: invoke Response.follow_all --- scrapy/http/response/text.py | 29 +++++++++++++---------------- 1 file changed, 13 insertions(+), 16 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 25e115bf9..f782f6217 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -201,22 +201,19 @@ class TextResponse(Response): urls.append(_url_from_selector(selector)) except ValueError: pass - return ( - self.follow( - url=url, - callback=callback, - method=method, - headers=headers, - body=body, - cookies=cookies, - meta=meta, - encoding=encoding, - priority=priority, - dont_filter=dont_filter, - errback=errback, - cb_kwargs=cb_kwargs, - ) - for url in urls + return super(TextResponse, self).follow_all( + urls=urls, + callback=callback, + method=method, + headers=headers, + body=body, + cookies=cookies, + meta=meta, + encoding=encoding, + priority=priority, + dont_filter=dont_filter, + errback=errback, + cb_kwargs=cb_kwargs, ) From 6df6b6dd6a5afd9172c03f6b9e8ce8e180867339 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Mon, 21 Oct 2019 03:56:45 -0300 Subject: [PATCH 024/313] Initializer -> __init__ --- scrapy/http/response/text.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index f782f6217..74017b5aa 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -167,9 +167,9 @@ class TextResponse(Response): # type: (...) -> Generator[Request, None, None] """ A generator that produces :class:`~.Request` instances to follow all - links in ``urls``. It accepts the same arguments as the :class:`~.Request` - initializer, except that each ``urls`` element does not need to be an absolute - URL, it can be any of the following: + links in ``urls``. It accepts the same arguments as the :class:`~.Request`'s + ``__init__`` method, except that each ``urls`` element does not need to be + an absolute URL, it can be any of the following: * a relative URL * a :class:`~scrapy.link.Link` object, e.g. the result of From 0fbd1ff4a91c68e552a3f919824c60437fc9a141 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 21 Oct 2019 14:06:45 +0200 Subject: [PATCH 025/313] =?UTF-8?q?constructor=20=E2=86=92=20=5F=5Finit=5F?= =?UTF-8?q?=5F=20method?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/topics/link-extractors.rst | 6 +++--- scrapy/linkextractors/lxmlhtml.py | 4 ++-- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index f9936a498..2119cb8f8 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -6,9 +6,9 @@ Link Extractors A link extractor is an object that extracts links from responses. -The constructor of :class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` -takes settings that determine which links may be extracted. -:class:`LxmlLinkExtractor.extract_links +The ``__init__`` method of +:class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` takes settings that +determine which links may be extracted. :class:`LxmlLinkExtractor.extract_links ` returns a list of matching :class:`scrapy.link.Link` objects from a :class:`~scrapy.http.Response` object. diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index 41091ba23..37003720f 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -120,8 +120,8 @@ class LxmlLinkExtractor(FilteringLinkExtractor): """Returns a list of :class:`~scrapy.link.Link` objects from the specified :class:`response `. - Only links that match the settings passed to the link extractor - constructor are returned. + Only links that match the settings passed to the ``__init__`` method of + the link extractor are returned. Duplicate links are omitted. """ From 058bdda0afe967f99a3ffd59b395566b063d9c3f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 14 Nov 2019 16:51:47 +0100 Subject: [PATCH 026/313] Improve the performance of the DOWNLOAD_DELAY test --- tests/test_crawl.py | 51 +++++++++++++++++++++++++++++++-------------- 1 file changed, 35 insertions(+), 16 deletions(-) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 3fc13eeb7..a524287eb 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -30,25 +30,44 @@ class CrawlTestCase(TestCase): self.assertEqual(len(crawler.spider.urls_visited), 11) # 10 + start_url @defer.inlineCallbacks - def test_delay(self): - # short to long delays - yield self._test_delay(0.2, False) - yield self._test_delay(1, False) - # randoms - yield self._test_delay(0.2, True) - yield self._test_delay(1, True) + def test_fixed_delay(self): + yield self._test_delay(total=3, delay=0.1) @defer.inlineCallbacks - def _test_delay(self, delay, randomize): - settings = {"DOWNLOAD_DELAY": delay, 'RANDOMIZE_DOWNLOAD_DELAY': randomize} + def test_randomized_delay(self): + yield self._test_delay(total=3, delay=0.1, randomize=True) + + @defer.inlineCallbacks + def _test_delay(self, total, delay, randomize=False): + crawl_kwargs = dict( + maxlatency=delay * 2, + mockserver=self.mockserver, + total=total, + ) + tolerance = (1 - (0.6 if randomize else 0.2)) + + settings = {"DOWNLOAD_DELAY": delay, + 'RANDOMIZE_DOWNLOAD_DELAY': randomize} crawler = CrawlerRunner(settings).create_crawler(FollowAllSpider) - yield crawler.crawl(maxlatency=delay * 2, mockserver=self.mockserver) - t = crawler.spider.times - totaltime = t[-1] - t[0] - avgd = totaltime / (len(t) - 1) - tolerance = 0.6 if randomize else 0.2 - self.assertTrue(avgd > delay * (1 - tolerance), - "download delay too small: %s" % avgd) + yield crawler.crawl(**crawl_kwargs) + times = crawler.spider.times + total_time = times[-1] - times[0] + average = total_time / (len(times) - 1) + self.assertTrue(average > delay * tolerance, + "download delay too small: %s" % average) + + # Ensure that the same test parameters would cause a failure if no + # download delay is set. Otherwise, it means we are using a combination + # of ``total`` and ``delay`` values that are too small for the test + # code above to have any meaning. + settings["DOWNLOAD_DELAY"] = 0 + crawler = CrawlerRunner(settings).create_crawler(FollowAllSpider) + yield crawler.crawl(**crawl_kwargs) + times = crawler.spider.times + total_time = times[-1] - times[0] + average = total_time / (len(times) - 1) + self.assertFalse(average > delay / tolerance, + "test total or delay values are too small") @defer.inlineCallbacks def test_timeout_success(self): From 6d1667d5b8086dd12539a185680f9e119c5fdc55 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 7 Nov 2019 18:32:33 +0100 Subject: [PATCH 027/313] Use the latest Python version to build the documentation --- .travis.yml | 2 +- tox.ini | 10 ++++++++-- 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/.travis.yml b/.travis.yml index 9f477e860..c9c64e990 100644 --- a/.travis.yml +++ b/.travis.yml @@ -26,7 +26,7 @@ matrix: - env: TOXENV=py38-extra-deps python: 3.8 - env: TOXENV=docs - python: 3.6 + python: 3.8 install: - | if [ "$TOXENV" = "pypy3" ]; then diff --git a/tox.ini b/tox.ini index fd75d18e2..195cc106a 100644 --- a/tox.ini +++ b/tox.ini @@ -6,6 +6,9 @@ [tox] envlist = py35 +[latest] +basepython = python3.8 + [testenv] deps = -ctests/constraints.txt @@ -63,14 +66,14 @@ commands = py.test {posargs:--durations=10 docs scrapy tests} [testenv:security] -basepython = python3.8 +basepython = {[latest]basepython} deps = bandit commands = bandit -r -c .bandit.yml {posargs:scrapy} [testenv:flake8] -basepython = python3.8 +basepython = {[latest]basepython} deps = {[testenv]deps} pytest-flake8 @@ -83,18 +86,21 @@ deps = -rdocs/requirements.txt [testenv:docs] +basepython = {[latest]basepython} changedir = {[docs]changedir} deps = {[docs]deps} commands = sphinx-build -W -b html . {envtmpdir}/html [testenv:docs-coverage] +basepython = {[latest]basepython} changedir = {[docs]changedir} deps = {[docs]deps} commands = sphinx-build -b coverage . {envtmpdir}/coverage [testenv:docs-links] +basepython = {[latest]basepython} changedir = {[docs]changedir} deps = {[docs]deps} commands = From e1af85619f63fe9312920f0dea7454bb76fcc23f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 18 Nov 2019 11:06:25 +0100 Subject: [PATCH 028/313] Add a configuration file for Read the Docs --- .readthedocs.yml | 7 +++++++ 1 file changed, 7 insertions(+) create mode 100644 .readthedocs.yml diff --git a/.readthedocs.yml b/.readthedocs.yml new file mode 100644 index 000000000..3c1c3e8be --- /dev/null +++ b/.readthedocs.yml @@ -0,0 +1,7 @@ +version: 2 +sphinx: + configuration: docs/conf.py +python: + version: 3.8 + install: + - requirements: docs/requirements.txt From fc3af54dbd5b7fdcbc44c82ba4ced528a104cd88 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Wed, 20 Nov 2019 07:57:09 +0100 Subject: [PATCH 029/313] Make tox configuration more user friendly --- .travis.yml | 5 ++-- docs/contributing.rst | 14 ++-------- tox.ini | 64 ++++++++++++++++++------------------------- 3 files changed, 31 insertions(+), 52 deletions(-) diff --git a/.travis.yml b/.travis.yml index 9f477e860..4e28d6f11 100644 --- a/.travis.yml +++ b/.travis.yml @@ -12,10 +12,9 @@ matrix: - env: TOXENV=flake8 python: 3.8 - env: TOXENV=pypy3 - python: 3.5 - env: TOXENV=py35 python: 3.5 - - env: TOXENV=py35-pinned + - env: TOXENV=pinned python: 3.5 - env: TOXENV=py36 python: 3.6 @@ -23,7 +22,7 @@ matrix: python: 3.7 - env: TOXENV=py38 python: 3.8 - - env: TOXENV=py38-extra-deps + - env: TOXENV=extra-deps python: 3.8 - env: TOXENV=docs python: 3.6 diff --git a/docs/contributing.rst b/docs/contributing.rst index f084bd23d..c4cb605ab 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -202,17 +202,9 @@ tests requires `tox`_. Running tests ------------- -Make sure you have a recent enough `tox`_ installation: +To run all tests:: - ``tox --version`` - -If your version is older than 1.7.0, please update it first: - - ``pip install -U tox`` - -To run all tests go to the root directory of Scrapy source code and run: - - ``tox`` + tox To run a specific test (say ``tests/test_loader.py``) use: @@ -227,7 +219,7 @@ environment name from ``tox.ini``. For example, to run the tests with Python You can also specify a comma-separated list of environmets, and use `tox’s parallel mode`_ to run the tests on multiple environments in parallel:: - tox -e py27,py36 -p auto + tox -e py37,py38 -p auto To pass command-line options to pytest_, add them after ``--`` in your call to tox_. Using ``--`` overrides the default positional arguments defined in diff --git a/tox.ini b/tox.ini index fd75d18e2..0b41f6cc3 100644 --- a/tox.ini +++ b/tox.ini @@ -4,7 +4,8 @@ # and then run "tox" from this directory. [tox] -envlist = py35 +envlist = security,flake8,py3 +minversion = 1.7.0 [testenv] deps = @@ -23,11 +24,28 @@ passenv = commands = py.test --cov=scrapy --cov-report= {posargs:--durations=10 docs scrapy tests} -[testenv:py35] -basepython = python3.5 +[testenv:security] +basepython = python3 +deps = + bandit +commands = + bandit -r -c .bandit.yml {posargs:scrapy} -[testenv:py35-pinned] -basepython = python3.5 +[testenv:flake8] +basepython = python3 +deps = + {[testenv]deps} + pytest-flake8 +commands = + py.test --flake8 {posargs:docs scrapy tests} + +[testenv:pypy3] +basepython = pypy3 +commands = + py.test {posargs:--durations=10 docs scrapy tests} + +[testenv:pinned] +basepython = python3 deps = -ctests/constraints.txt cryptography==2.0 @@ -48,34 +66,11 @@ deps = botocore==1.3.23 Pillow==3.4.2 -[testenv:py36] -basepython = python3.6 - -[testenv:py37] -basepython = python3.7 - -[testenv:py38] -basepython = python3.8 - -[testenv:pypy3] -basepython = pypy3 -commands = - py.test {posargs:--durations=10 docs scrapy tests} - -[testenv:security] -basepython = python3.8 -deps = - bandit -commands = - bandit -r -c .bandit.yml {posargs:scrapy} - -[testenv:flake8] -basepython = python3.8 +[testenv:extra-deps] deps = {[testenv]deps} - pytest-flake8 -commands = - py.test --flake8 {posargs:docs scrapy tests} + reppy + robotexclusionrulesparser [docs] changedir = docs @@ -99,10 +94,3 @@ changedir = {[docs]changedir} deps = {[docs]deps} commands = sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck - -[testenv:py38-extra-deps] -basepython = python3.8 -deps = - {[testenv]deps} - reppy - robotexclusionrulesparser From e6c5292a7c4391642897ee31ea2ded889b3b88dc Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 20 Nov 2019 09:29:55 -0300 Subject: [PATCH 030/313] Response.follow_all: Specific exception for invalid selectors --- scrapy/http/response/text.py | 31 ++++++++++++++++++------------- 1 file changed, 18 insertions(+), 13 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 74017b5aa..5110b4bd4 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -5,17 +5,18 @@ discovering (through HTTP headers) to base Response class. See documentation in docs/topics/request-response.rst """ -import six -from six.moves.urllib.parse import urljoin +from contextlib import suppress import parsel -from w3lib.encoding import html_to_unicode, resolve_encoding, \ - html_body_declared_encoding, http_content_type_encoding +import six +from six.moves.urllib.parse import urljoin +from w3lib.encoding import (html_body_declared_encoding, html_to_unicode, + http_content_type_encoding, resolve_encoding) from w3lib.html import strip_html5_whitespace from scrapy.http.response import Response -from scrapy.utils.response import get_base_url from scrapy.utils.python import memoizemethod_noargs, to_native_str +from scrapy.utils.response import get_base_url class TextResponse(Response): @@ -197,10 +198,8 @@ class TextResponse(Response): selector_list = self.xpath(xpath) urls = [] for selector in selector_list: - try: + with suppress(_InvalidSelector): urls.append(_url_from_selector(selector)) - except ValueError: - pass return super(TextResponse, self).follow_all( urls=urls, callback=callback, @@ -217,18 +216,24 @@ class TextResponse(Response): ) +class _InvalidSelector(ValueError): + """ + Raised when a URL cannot be obtained from a Selector + """ + + def _url_from_selector(sel): # type: (parsel.Selector) -> str if isinstance(sel.root, six.string_types): # e.g. ::attr(href) result return strip_html5_whitespace(sel.root) if not hasattr(sel.root, 'tag'): - raise ValueError("Unsupported selector: %s" % sel) + raise _InvalidSelector("Unsupported selector: %s" % sel) if sel.root.tag not in ('a', 'link'): - raise ValueError("Only and elements are supported; got <%s>" % - sel.root.tag) + raise _InvalidSelector("Only and elements are supported; got <%s>" % + sel.root.tag) href = sel.root.get('href') if href is None: - raise ValueError("<%s> element has no href attribute: %s" % - (sel.root.tag, sel)) + raise _InvalidSelector("<%s> element has no href attribute: %s" % + (sel.root.tag, sel)) return strip_html5_whitespace(href) From b602c61e1cc00e04ce435078644d5ea0c0249903 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 20 Nov 2019 09:38:54 -0300 Subject: [PATCH 031/313] [Test] Rename outdated sample files --- .../{sgml_linkextractor.html => linkextractor.html} | 0 ..._linkextractor_no_href.html => linkextractor_no_href.html} | 0 tests/test_http_response.py | 4 ++-- 3 files changed, 2 insertions(+), 2 deletions(-) rename tests/sample_data/link_extractor/{sgml_linkextractor.html => linkextractor.html} (100%) rename tests/sample_data/link_extractor/{sgml_linkextractor_no_href.html => linkextractor_no_href.html} (100%) diff --git a/tests/sample_data/link_extractor/sgml_linkextractor.html b/tests/sample_data/link_extractor/linkextractor.html similarity index 100% rename from tests/sample_data/link_extractor/sgml_linkextractor.html rename to tests/sample_data/link_extractor/linkextractor.html diff --git a/tests/sample_data/link_extractor/sgml_linkextractor_no_href.html b/tests/sample_data/link_extractor/linkextractor_no_href.html similarity index 100% rename from tests/sample_data/link_extractor/sgml_linkextractor_no_href.html rename to tests/sample_data/link_extractor/linkextractor_no_href.html diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 6d3c5cb9d..0ae1612b5 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -250,12 +250,12 @@ class BaseResponseTest(unittest.TestCase): yield req def _links_response(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') + body = get_testdata('link_extractor', 'linkextractor.html') resp = self.response_class('http://example.com/index', body=body) return resp def _links_response_no_href(self): - body = get_testdata('link_extractor', 'sgml_linkextractor_no_href.html') + body = get_testdata('link_extractor', 'linkextractor_no_href.html') resp = self.response_class('http://example.com/index', body=body) return resp From 6f4e84ecf95feabe15fda0819bc428f8e0d4d340 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 20 Nov 2019 09:55:15 -0300 Subject: [PATCH 032/313] PEP8 adjustments for scrapy.http.response module --- scrapy/http/response/__init__.py | 6 ++++-- scrapy/http/response/html.py | 1 + scrapy/http/response/text.py | 2 ++ scrapy/http/response/xml.py | 1 + 4 files changed, 8 insertions(+), 2 deletions(-) diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index 79a8d0ca0..b9e638551 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -4,6 +4,8 @@ responses in Scrapy. See documentation in docs/topics/request-response.rst """ +from typing import Generator + from six.moves.urllib.parse import urljoin from scrapy.http.request import Request @@ -41,8 +43,8 @@ class Response(object_ref): if isinstance(url, str): self._url = url else: - raise TypeError('%s url must be str, got %s:' % (type(self).__name__, - type(url).__name__)) + raise TypeError('%s url must be str, got %s:' % + (type(self).__name__, type(url).__name__)) url = property(_get_url, obsolete_setter(_set_url, 'url')) diff --git a/scrapy/http/response/html.py b/scrapy/http/response/html.py index bd3559fbb..7eed052c2 100644 --- a/scrapy/http/response/html.py +++ b/scrapy/http/response/html.py @@ -7,5 +7,6 @@ See documentation in docs/topics/request-response.rst from scrapy.http.response.text import TextResponse + class HtmlResponse(TextResponse): pass diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index b97420345..6acf1026f 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -6,6 +6,7 @@ See documentation in docs/topics/request-response.rst """ from contextlib import suppress +from typing import Generator import parsel import six @@ -14,6 +15,7 @@ from w3lib.encoding import (html_body_declared_encoding, html_to_unicode, http_content_type_encoding, resolve_encoding) from w3lib.html import strip_html5_whitespace +from scrapy.http import Request from scrapy.http.response import Response from scrapy.utils.python import memoizemethod_noargs, to_unicode from scrapy.utils.response import get_base_url diff --git a/scrapy/http/response/xml.py b/scrapy/http/response/xml.py index 1df33fee5..abf474a2f 100644 --- a/scrapy/http/response/xml.py +++ b/scrapy/http/response/xml.py @@ -7,5 +7,6 @@ See documentation in docs/topics/request-response.rst from scrapy.http.response.text import TextResponse + class XmlResponse(TextResponse): pass From 6781d2f5b261305a4f6aa41a2b485a2e5cc7bc76 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 20 Nov 2019 09:58:25 -0300 Subject: [PATCH 033/313] Update sample file references --- tests/test_linkextractors.py | 2 +- tests/test_linkextractors_deprecated.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index 57ef1694a..0b94f937f 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -16,7 +16,7 @@ class Base: escapes_whitespace = False def setUp(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') + body = get_testdata('link_extractor', 'linkextractor.html') self.response = HtmlResponse(url='http://example.com/index', body=body) def test_urls_type(self): diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py index 1366971be..388ed6ad4 100644 --- a/tests/test_linkextractors_deprecated.py +++ b/tests/test_linkextractors_deprecated.py @@ -111,7 +111,7 @@ class BaseSgmlLinkExtractorTestCase(unittest.TestCase): class HtmlParserLinkExtractorTestCase(unittest.TestCase): def setUp(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') + body = get_testdata('link_extractor', 'linkextractor.html') self.response = HtmlResponse(url='http://example.com/index', body=body) def test_extraction(self): @@ -183,7 +183,7 @@ class RegexLinkExtractorTestCase(unittest.TestCase): # than it should be. def setUp(self): - body = get_testdata('link_extractor', 'sgml_linkextractor.html') + body = get_testdata('link_extractor', 'linkextractor.html') self.response = HtmlResponse(url='http://example.com/index', body=body) def test_extraction(self): From 7a7d13b1122dac397ee0bb8edd4e6fd61665e232 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Sat, 23 Nov 2019 19:04:02 -0300 Subject: [PATCH 034/313] Rename LogFormatter.error to item_error --- scrapy/core/scraper.py | 2 +- scrapy/logformatter.py | 6 +++--- tests/test_logformatter.py | 4 ++-- 3 files changed, 6 insertions(+), 6 deletions(-) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index db463f989..c5bb48ea6 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -231,7 +231,7 @@ class Scraper(object): signal=signals.item_dropped, item=item, response=response, spider=spider, exception=output.value) else: - logkws = self.logformatter.error(item, ex, response, spider) + logkws = self.logformatter.item_error(item, ex, response, spider) logger.log(*logformatter_adapter(logkws), extra={'spider': spider}, exc_info=failure_to_exc_info(output)) return self.signals.send_catch_log_deferred( diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 5189d7cfa..79c752da4 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -8,7 +8,7 @@ from scrapy.utils.request import referer_str SCRAPEDMSG = u"Scraped from %(src)s" + os.linesep + "%(item)s" DROPPEDMSG = u"Dropped: %(exception)s" + os.linesep + "%(item)s" CRAWLEDMSG = u"Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" -ERRORMSG = u"'Error processing %(item)s'" +ITEMERRORMSG = u"'Error processing %(item)s'" class LogFormatter(object): @@ -93,11 +93,11 @@ class LogFormatter(object): } } - def error(self, item, exception, response, spider): + def item_error(self, item, exception, response, spider): """Logs a message when an item causes an error while it is passing through the item pipeline.""" return { 'level': logging.ERROR, - 'msg': ERRORMSG, + 'msg': ITEMERRORMSG, 'args': { 'item': item, } diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index d0b23a8c4..f2f8c0464 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -63,13 +63,13 @@ class LogFormatterTestCase(unittest.TestCase): assert all(isinstance(x, six.text_type) for x in lines) self.assertEqual(lines, [u"Dropped: \u2018", '{}']) - def test_error(self): + def test_item_error(self): # In practice, the complete traceback is shown by passing the # 'exc_info' argument to the logging function item = {'key': 'value'} exception = Exception() response = Response("http://www.example.com") - logkws = self.formatter.error(item, exception, response, self.spider) + logkws = self.formatter.item_error(item, exception, response, self.spider) logline = logkws['msg'] % logkws['args'] self.assertEqual(logline, u"'Error processing {'key': 'value'}'") From facb9265421ead8afb839323af2e18f81dda560b Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Sat, 23 Nov 2019 19:16:41 -0300 Subject: [PATCH 035/313] Remove quotes from item_error message --- scrapy/logformatter.py | 8 ++++---- tests/test_logformatter.py | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 79c752da4..9e038160f 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -5,10 +5,10 @@ from twisted.python.failure import Failure from scrapy.utils.request import referer_str -SCRAPEDMSG = u"Scraped from %(src)s" + os.linesep + "%(item)s" -DROPPEDMSG = u"Dropped: %(exception)s" + os.linesep + "%(item)s" -CRAWLEDMSG = u"Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" -ITEMERRORMSG = u"'Error processing %(item)s'" +SCRAPEDMSG = "Scraped from %(src)s" + os.linesep + "%(item)s" +DROPPEDMSG = "Dropped: %(exception)s" + os.linesep + "%(item)s" +CRAWLEDMSG = "Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" +ITEMERRORMSG = "Error processing %(item)s" class LogFormatter(object): diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index f2f8c0464..990927f71 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -71,7 +71,7 @@ class LogFormatterTestCase(unittest.TestCase): response = Response("http://www.example.com") logkws = self.formatter.item_error(item, exception, response, self.spider) logline = logkws['msg'] % logkws['args'] - self.assertEqual(logline, u"'Error processing {'key': 'value'}'") + self.assertEqual(logline, u"Error processing {'key': 'value'}") def test_scraped(self): item = CustomItem() From 4756e7c587880997a54d9abf94b9a4c0b5bab71c Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Sat, 23 Nov 2019 19:33:29 -0300 Subject: [PATCH 036/313] LogFormatter.spider_error --- scrapy/core/scraper.py | 8 ++++---- scrapy/logformatter.py | 12 ++++++++++++ tests/test_logformatter.py | 14 ++++++++++++++ 3 files changed, 30 insertions(+), 4 deletions(-) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index c5bb48ea6..21820e988 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -16,7 +16,7 @@ from scrapy import signals from scrapy.http import Request, Response from scrapy.item import BaseItem from scrapy.core.spidermw import SpiderMiddlewareManager -from scrapy.utils.request import referer_str + logger = logging.getLogger(__name__) @@ -152,9 +152,9 @@ class Scraper(object): if isinstance(exc, CloseSpider): self.crawler.engine.close_spider(spider, exc.reason or 'cancelled') return - logger.error( - "Spider error processing %(request)s (referer: %(referer)s)", - {'request': request, 'referer': referer_str(request)}, + logkws = self.logformatter.spider_error(_failure, request, response, spider) + logger.log( + *logformatter_adapter(logkws), exc_info=failure_to_exc_info(_failure), extra={'spider': spider} ) diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 9e038160f..d87f685d5 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -9,6 +9,7 @@ SCRAPEDMSG = "Scraped from %(src)s" + os.linesep + "%(item)s" DROPPEDMSG = "Dropped: %(exception)s" + os.linesep + "%(item)s" CRAWLEDMSG = "Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" ITEMERRORMSG = "Error processing %(item)s" +SPIDERERRORMSG = "Spider error processing %(request)s (referer: %(referer)s)" class LogFormatter(object): @@ -103,6 +104,17 @@ class LogFormatter(object): } } + def spider_error(self, failure, request, response, spider): + """Logs an error message from a spider.""" + return { + 'level': logging.ERROR, + 'msg': SPIDERERRORMSG, + 'args': { + 'request': request, + 'referer': referer_str(request), + } + } + @classmethod def from_crawler(cls, crawler): return cls() diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index 990927f71..47d2747c2 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -2,6 +2,7 @@ import unittest from testfixtures import LogCapture from twisted.internet import defer +from twisted.python.failure import Failure from twisted.trial.unittest import TestCase as TwistedTestCase import six @@ -73,6 +74,19 @@ class LogFormatterTestCase(unittest.TestCase): logline = logkws['msg'] % logkws['args'] self.assertEqual(logline, u"Error processing {'key': 'value'}") + def test_spider_error(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + failure = Failure(Exception()) + request = Request("http://www.example.com", headers={'Referer': 'http://example.org'}) + response = Response("http://www.example.com", request=request) + logkws = self.formatter.spider_error(failure, request, response, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual( + logline, + "Spider error processing (referer: http://example.org)" + ) + def test_scraped(self): item = CustomItem() item['name'] = u'\xa3' From 03af8885ff475dc47a3de89517b1a5d627bd49c4 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Sat, 23 Nov 2019 20:02:44 -0300 Subject: [PATCH 037/313] LogFormatter.download_error --- scrapy/core/scraper.py | 22 +++++++++++++--------- scrapy/logformatter.py | 16 ++++++++++++++++ tests/test_logformatter.py | 18 ++++++++++++++++++ 3 files changed, 47 insertions(+), 9 deletions(-) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 21820e988..427969f30 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -200,19 +200,23 @@ class Scraper(object): """Log and silence errors that come from the engine (typically download errors that got propagated thru here) """ - if (isinstance(download_failure, Failure) and - not download_failure.check(IgnoreRequest)): + if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest): if download_failure.frames: - logger.error('Error downloading %(request)s', - {'request': request}, - exc_info=failure_to_exc_info(download_failure), - extra={'spider': spider}) + logkws = self.logformatter.download_error(download_failure, request, spider) + logger.log( + *logformatter_adapter(logkws), + extra={'spider': spider}, + exc_info=failure_to_exc_info(download_failure), + ) else: errmsg = download_failure.getErrorMessage() if errmsg: - logger.error('Error downloading %(request)s: %(errmsg)s', - {'request': request, 'errmsg': errmsg}, - extra={'spider': spider}) + logkws = self.logformatter.download_error( + download_failure, request, spider, errmsg) + logger.log( + *logformatter_adapter(logkws), + extra={'spider': spider}, + ) if spider_failure is not download_failure: return spider_failure diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index d87f685d5..99bd5cfac 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -10,6 +10,8 @@ DROPPEDMSG = "Dropped: %(exception)s" + os.linesep + "%(item)s" CRAWLEDMSG = "Crawled (%(status)s) %(request)s%(request_flags)s (referer: %(referer)s)%(response_flags)s" ITEMERRORMSG = "Error processing %(item)s" SPIDERERRORMSG = "Spider error processing %(request)s (referer: %(referer)s)" +DOWNLOADERRORMSG_SHORT = "Error downloading %(request)s" +DOWNLOADERRORMSG_LONG = "Error downloading %(request)s: %(errmsg)s" class LogFormatter(object): @@ -115,6 +117,20 @@ class LogFormatter(object): } } + def download_error(self, failure, request, spider, errmsg=None): + """Logs a download error message from a spider (typically coming from the engine).""" + args = {'request': request} + if errmsg: + msg = DOWNLOADERRORMSG_LONG + args['errmsg'] = errmsg + else: + msg = DOWNLOADERRORMSG_SHORT + return { + 'level': logging.ERROR, + 'msg': msg, + 'args': args, + } + @classmethod def from_crawler(cls, crawler): return cls() diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index 47d2747c2..697ac1d15 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -87,6 +87,24 @@ class LogFormatterTestCase(unittest.TestCase): "Spider error processing (referer: http://example.org)" ) + def test_download_error_short(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + failure = Failure(Exception()) + request = Request("http://www.example.com") + logkws = self.formatter.download_error(failure, request, self.spider) + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, "Error downloading ") + + def test_download_error_long(self): + # In practice, the complete traceback is shown by passing the + # 'exc_info' argument to the logging function + failure = Failure(Exception()) + request = Request("http://www.example.com") + logkws = self.formatter.download_error(failure, request, self.spider, "Some message") + logline = logkws['msg'] % logkws['args'] + self.assertEqual(logline, "Error downloading : Some message") + def test_scraped(self): item = CustomItem() item['name'] = u'\xa3' From 54b056c4be045099f9f872c5455c003d68b578a7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 25 Nov 2019 12:13:31 +0100 Subject: [PATCH 038/313] Make developer-tools doctests pass --- docs/_tests/quotes.html | 281 ++++++++++++++++++++++++++++++++ docs/topics/developer-tools.rst | 40 +++-- pytest.ini | 1 - 3 files changed, 308 insertions(+), 14 deletions(-) create mode 100644 docs/_tests/quotes.html diff --git a/docs/_tests/quotes.html b/docs/_tests/quotes.html new file mode 100644 index 000000000..71aff8847 --- /dev/null +++ b/docs/_tests/quotes.html @@ -0,0 +1,281 @@ + + + + + Quotes to Scrape + + + + +
+
+ +
+

+ + Login + +

+
+
+ + +
+
+ +
+ “The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.” + by + (about) + +
+ Tags: + + + change + + deep-thoughts + + thinking + + world + +
+
+ +
+ “It is our choices, Harry, that show what we truly are, far more than our abilities.” + by + (about) + +
+ Tags: + + + abilities + + choices + +
+
+ +
+ “There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.” + by + (about) + +
+ Tags: + + + inspirational + + life + + live + + miracle + + miracles + +
+
+ +
+ “The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.” + by + (about) + +
+ Tags: + + + aliteracy + + books + + classic + + humor + +
+
+ +
+ “Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.” + by + (about) + +
+ Tags: + + + be-yourself + + inspirational + +
+
+ +
+ “Try not to become a man of success. Rather become a man of value.” + by + (about) + +
+ Tags: + + + adulthood + + success + + value + +
+
+ +
+ “It is better to be hated for what you are than to be loved for what you are not.” + by + (about) + +
+ Tags: + + + life + + love + +
+
+ +
+ “I have not failed. I've just found 10,000 ways that won't work.” + by + (about) + +
+ Tags: + + + edison + + failure + + inspirational + + paraphrased + +
+
+ +
+ “A woman is like a tea bag; you never know how strong it is until it's in hot water.” + by + (about) + + +
+ +
+ “A day without sunshine is like, you know, night.” + by + (about) + +
+ Tags: + + + humor + + obvious + + simile + +
+
+ + +
+
+ +

Top Ten tags

+ + + love + + + + inspirational + + + + life + + + + humor + + + + books + + + + reading + + + + friendship + + + + friends + + + + truth + + + + simile + + + +
+
+ +
+ + + \ No newline at end of file diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index bf14643be..e67ce55f9 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -39,7 +39,7 @@ Therefore, you should keep in mind the following things: .. _topics-inspector: Inspecting a website -=================================== +==================== By far the most handy feature of the Developer Tools is the `Inspector` feature, which allows you to inspect the underlying HTML code of @@ -79,13 +79,23 @@ sections and tags of a webpage, which greatly improves readability. You can expand and collapse a tag by clicking on the arrow in front of it or by double clicking directly on the tag. If we expand the ``span`` tag with the ``class= "text"`` we will see the quote-text we clicked on. The `Inspector` lets you -copy XPaths to selected elements. Let's try it out: Right-click on the ``span`` -tag, select ``Copy > XPath`` and paste it in the scrapy shell like so:: +copy XPaths to selected elements. Let's try it out. + +First open the Scrapy shell at http://quotes.toscrape.com/ in a terminal: + +.. code-block:: none $ scrapy shell "http://quotes.toscrape.com/" - (...) - >>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall() - ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”] + +Then, back to your web browser, right-click on the ``span`` tag, select +``Copy > XPath`` and paste it in the Scrapy shell like so: + +.. invisible-code-block: python + + response = load_response('http://quotes.toscrape.com/', 'quotes.html') + +>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall() +['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'] Adding ``text()`` at the end we are able to extract the first quote with this basic selector. But this XPath is not really that clever. All it does is @@ -112,13 +122,13 @@ see each quote: With this knowledge we can refine our XPath: Instead of a path to follow, we'll simply select all ``span`` tags with the ``class="text"`` by using -the `has-class-extension`_:: +the `has-class-extension`_: - >>> response.xpath('//span[has-class("text")]/text()').getall() - ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”, - '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', - '“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', - (...)] +>>> response.xpath('//span[has-class("text")]/text()').getall() +['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', +'“It is our choices, Harry, that show what we truly are, far more than our abilities.”', +'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', +...] And with one simple, cleverer XPath we are able to extract all quotes from the page. We could have constructed a loop over our first XPath to increase @@ -159,7 +169,11 @@ The page is quite similar to the basic `quotes.toscrape.com`_-page, but instead of the above-mentioned ``Next`` button, the page automatically loads new quotes when you scroll to the bottom. We could go ahead and try out different XPaths directly, but instead -we'll check another quite useful command from the scrapy shell:: +we'll check another quite useful command from the scrapy shell: + +.. skip: next + +.. code-block:: none $ scrapy shell "quotes.toscrape.com/scroll" (...) diff --git a/pytest.ini b/pytest.ini index 33c34b8e8..0e830076d 100644 --- a/pytest.ini +++ b/pytest.ini @@ -8,7 +8,6 @@ addopts = --ignore=docs/_ext --ignore=docs/conf.py --ignore=docs/news.rst - --ignore=docs/topics/developer-tools.rst --ignore=docs/topics/dynamic-content.rst --ignore=docs/topics/items.rst --ignore=docs/topics/leaks.rst From ed1e577610b77e92f9833ca96755b9f0727085f5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 25 Nov 2019 13:35:29 +0100 Subject: [PATCH 039/313] Use super().__init__ in BaseItemExporter subclasses --- scrapy/exporters.py | 31 ++++++++++++++++--------------- 1 file changed, 16 insertions(+), 15 deletions(-) diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 3defafd60..0c28c2d7f 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -24,8 +24,9 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter', class BaseItemExporter(object): - def __init__(self, **kwargs): - self._configure(kwargs) + def __init__(self, dont_fail=False, **kwargs): + self._kwargs = kwargs + self._configure(kwargs, dont_fail=dont_fail) def _configure(self, options, dont_fail=False): """Configure the exporter by poping options from the ``options`` dict. @@ -82,10 +83,10 @@ class BaseItemExporter(object): class JsonLinesItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): - self._configure(kwargs, dont_fail=True) + super().__init__(dont_fail=True, **kwargs) self.file = file - kwargs.setdefault('ensure_ascii', not self.encoding) - self.encoder = ScrapyJSONEncoder(**kwargs) + self._kwargs.setdefault('ensure_ascii', not self.encoding) + self.encoder = ScrapyJSONEncoder(**self._kwargs) def export_item(self, item): itemdict = dict(self._get_serialized_fields(item)) @@ -96,15 +97,15 @@ class JsonLinesItemExporter(BaseItemExporter): class JsonItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): - self._configure(kwargs, dont_fail=True) + super().__init__(dont_fail=True, **kwargs) self.file = file # there is a small difference between the behaviour or JsonItemExporter.indent # and ScrapyJSONEncoder.indent. ScrapyJSONEncoder.indent=None is needed to prevent # the addition of newlines everywhere json_indent = self.indent if self.indent is not None and self.indent > 0 else None - kwargs.setdefault('indent', json_indent) - kwargs.setdefault('ensure_ascii', not self.encoding) - self.encoder = ScrapyJSONEncoder(**kwargs) + self._kwargs.setdefault('indent', json_indent) + self._kwargs.setdefault('ensure_ascii', not self.encoding) + self.encoder = ScrapyJSONEncoder(**self._kwargs) self.first_item = True def _beautify_newline(self): @@ -135,7 +136,7 @@ class XmlItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): self.item_element = kwargs.pop('item_element', 'item') self.root_element = kwargs.pop('root_element', 'items') - self._configure(kwargs) + super().__init__(**kwargs) if not self.encoding: self.encoding = 'utf-8' self.xg = XMLGenerator(file, encoding=self.encoding) @@ -191,7 +192,7 @@ class XmlItemExporter(BaseItemExporter): class CsvItemExporter(BaseItemExporter): def __init__(self, file, include_headers_line=True, join_multivalued=',', **kwargs): - self._configure(kwargs, dont_fail=True) + super().__init__(dont_fail=True, **kwargs) if not self.encoding: self.encoding = 'utf-8' self.include_headers_line = include_headers_line @@ -202,7 +203,7 @@ class CsvItemExporter(BaseItemExporter): encoding=self.encoding, newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034 ) - self.csv_writer = csv.writer(self.stream, **kwargs) + self.csv_writer = csv.writer(self.stream, **self._kwargs) self._headers_not_written = True self._join_multivalued = join_multivalued @@ -251,7 +252,7 @@ class CsvItemExporter(BaseItemExporter): class PickleItemExporter(BaseItemExporter): def __init__(self, file, protocol=2, **kwargs): - self._configure(kwargs) + super().__init__(**kwargs) self.file = file self.protocol = protocol @@ -270,7 +271,7 @@ class MarshalItemExporter(BaseItemExporter): """ def __init__(self, file, **kwargs): - self._configure(kwargs) + super().__init__(**kwargs) self.file = file def export_item(self, item): @@ -280,7 +281,7 @@ class MarshalItemExporter(BaseItemExporter): class PprintItemExporter(BaseItemExporter): def __init__(self, file, **kwargs): - self._configure(kwargs) + super().__init__(**kwargs) self.file = file def export_item(self, item): From dd12f5fdcd5ce2c63e9a5d3626031f5d93c1628e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 25 Nov 2019 15:59:59 +0100 Subject: [PATCH 040/313] Use Response.follow_all in the documentation where appropiate --- docs/intro/tutorial.rst | 26 +++++++++++++------------- 1 file changed, 13 insertions(+), 13 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 6b15a5fbd..a2775e0bb 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -625,12 +625,12 @@ attribute automatically. So the code can be shortened further:: for a in response.css('li.next a'): yield response.follow(a, callback=self.parse) -.. note:: +To create multiple requests from an iterable, you can use +:meth:`response.follow_all ` instead:: + + links = response.css('li.next a') + yield from response.follow_all(links, callback=self.parse) - ``response.follow(response.css('li.next a'))`` is not valid because - ``response.css`` returns a list-like object with selectors for all results, - not a single selector. A ``for`` loop like in the example above, or - ``response.follow(response.css('li.next a')[0])`` is fine. More examples and patterns -------------------------- @@ -647,13 +647,11 @@ this time for scraping author information:: start_urls = ['http://quotes.toscrape.com/'] def parse(self, response): - # follow links to author pages - for href in response.css('.author + a::attr(href)'): - yield response.follow(href, self.parse_author) + author_page_links = response.css('.author + a') + yield from response.follow_all(author_links, self.parse_author) - # follow pagination links - for href in response.css('li.next a::attr(href)'): - yield response.follow(href, self.parse) + pagination_links = response.css('li.next a') + yield from response.follow_all(pagination_links, self.parse) def parse_author(self, response): def extract_with_css(query): @@ -669,8 +667,10 @@ This spider will start from the main page, it will follow all the links to the authors pages calling the ``parse_author`` callback for each of them, and also the pagination links with the ``parse`` callback as we saw before. -Here we're passing callbacks to ``response.follow`` as positional arguments -to make the code shorter; it also works for ``scrapy.Request``. +Here we're passing callbacks to +:meth:`response.follow_all ` as positional +arguments to make the code shorter; it also works for +:class:`~scrapy.http.Request`. The ``parse_author`` callback defines a helper function to extract and cleanup the data from a CSS query and yields the Python dict with the author data. From 63546cbf3e02380819732441ad55d95725dca7c5 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 27 Nov 2019 22:42:52 +0500 Subject: [PATCH 041/313] Deprecate the HTTPS proxy noconnect mode. --- scrapy/core/downloader/handlers/http11.py | 7 ++++++ tests/test_downloader_handlers.py | 10 --------- tests/test_proxy_connect.py | 26 ----------------------- 3 files changed, 7 insertions(+), 36 deletions(-) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 7d917cb74..782eca89e 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -16,6 +16,7 @@ from twisted.web.http import _DataLoss, PotentialDataLoss from twisted.web.client import Agent, ResponseDone, HTTPConnectionPool, ResponseFailed, URI from twisted.internet.endpoints import TCP4ClientEndpoint +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Headers from scrapy.responsetypes import responsetypes from scrapy.core.downloader.webclient import _parse @@ -285,6 +286,12 @@ class ScrapyAgent(object): scheme = _parse(request.url)[0] proxyHost = to_unicode(proxyHost) omitConnectTunnel = b'noconnect' in proxyParams + if omitConnectTunnel: + warnings.warn("Using HTTPS proxies in the noconnect mode is deprecated. " + "If you use Crawlera, it doesn't require this mode anymore, " + "so you should update scrapy-crawlera to 1.3.0+ " + "and remove '?noconnect' from the Crawlera URL.", + ScrapyDeprecationWarning) if scheme == b'https' and not omitConnectTunnel: proxyAuth = request.headers.get(b'Proxy-Authorization', None) proxyConf = (proxyHost, proxyPort, proxyAuth) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 60124b93f..45d4aa952 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -687,16 +687,6 @@ class HttpProxyTestCase(unittest.TestCase): request = Request('http://example.com', meta={'proxy': http_proxy}) return self.download_request(request, Spider('foo')).addCallback(_test) - def test_download_with_proxy_https_noconnect(self): - def _test(response): - self.assertEqual(response.status, 200) - self.assertEqual(response.url, request.url) - self.assertEqual(response.body, b'https://example.com') - - http_proxy = '%s?noconnect' % self.getURL('') - request = Request('https://example.com', meta={'proxy': http_proxy}) - return self.download_request(request, Spider('foo')).addCallback(_test) - def test_download_without_proxy(self): def _test(response): self.assertEqual(response.status, 200) diff --git a/tests/test_proxy_connect.py b/tests/test_proxy_connect.py index 277455751..05d371b65 100644 --- a/tests/test_proxy_connect.py +++ b/tests/test_proxy_connect.py @@ -108,32 +108,6 @@ class ProxyConnectTestCase(TestCase): echo = json.loads(crawler.spider.meta['responses'][0].text) self.assertTrue('Proxy-Authorization' not in echo['headers']) - # The noconnect mode isn't supported by the current mitmproxy, it returns - # "Invalid request scheme: https" as it doesn't seem to support full URLs in GET at all, - # and it's not clear what behavior is intended by Scrapy and by mitmproxy here. - # https://github.com/mitmproxy/mitmproxy/issues/848 may be related. - # The Scrapy noconnect mode was required, at least in the past, to work with Crawlera, - # and https://github.com/scrapy-plugins/scrapy-crawlera/pull/44 seems to be related. - - @pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests') - @defer.inlineCallbacks - def test_https_noconnect(self): - proxy = os.environ['https_proxy'] - os.environ['https_proxy'] = proxy + '?noconnect' - crawler = get_crawler(SimpleSpider) - with LogCapture() as l: - yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True)) - self._assert_got_response_code(200, l) - - @pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests') - @defer.inlineCallbacks - def test_https_noconnect_auth_error(self): - os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy']) + '?noconnect' - crawler = get_crawler(SimpleSpider) - with LogCapture() as l: - yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True)) - self._assert_got_response_code(407, l) - def _assert_got_response_code(self, code, log): print(log) self.assertEqual(str(log).count('Crawled (%d)' % code), 1) From 17e648182332a1d231383c7416e08e89280cb1d0 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 27 Nov 2019 18:42:42 -0300 Subject: [PATCH 042/313] [Docs] Fix Twisted links --- docs/conf.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/conf.py b/docs/conf.py index eab366efd..a79f3a8cb 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -279,7 +279,7 @@ intersphinx_mapping = { 'python': ('https://docs.python.org/3', None), 'sphinx': ('https://www.sphinx-doc.org/en/master', None), 'tox': ('https://tox.readthedocs.io/en/latest', None), - 'twisted': ('https://twistedmatrix.com/documents/current', None), + 'twisted': ('https://twistedmatrix.com/documents/current/api', None), } From 048cd74ae594f449ba97d07c927d7640f32a6770 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 27 Nov 2019 19:16:18 -0300 Subject: [PATCH 043/313] Add separate mapping for Twisted API docs --- docs/conf.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/docs/conf.py b/docs/conf.py index a79f3a8cb..40e69c8ac 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -279,7 +279,8 @@ intersphinx_mapping = { 'python': ('https://docs.python.org/3', None), 'sphinx': ('https://www.sphinx-doc.org/en/master', None), 'tox': ('https://tox.readthedocs.io/en/latest', None), - 'twisted': ('https://twistedmatrix.com/documents/current/api', None), + 'twisted': ('https://twistedmatrix.com/documents/current', None), + 'twistedapi': ('https://twistedmatrix.com/documents/current/api', None), } From 5980b0f2840f51f8f9c7c3a266be93527999dd11 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Mon, 2 Dec 2019 16:47:44 +0100 Subject: [PATCH 044/313] =?UTF-8?q?Don=E2=80=99t=20use=20follow=5Fall=20wh?= =?UTF-8?q?ere=20a=20single=20item=20is=20expected=20(#4)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/intro/tutorial.rst | 11 +++++------ 1 file changed, 5 insertions(+), 6 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index a2775e0bb..2f97017fc 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -616,20 +616,19 @@ instance; you still have to yield this Request. You can also pass a selector to ``response.follow`` instead of a string; this selector should extract necessary attributes:: - for href in response.css('li.next a::attr(href)'): - yield response.follow(href, callback=self.parse) + href = response.css('li.next a::attr(href)')[0] + yield response.follow(href, callback=self.parse) For ```` elements there is a shortcut: ``response.follow`` uses their href attribute automatically. So the code can be shortened further:: - for a in response.css('li.next a'): - yield response.follow(a, callback=self.parse) + a = response.css('li.next a')[0] + yield response.follow(a, callback=self.parse) To create multiple requests from an iterable, you can use :meth:`response.follow_all ` instead:: - links = response.css('li.next a') - yield from response.follow_all(links, callback=self.parse) + yield from response.follow_all(response.css('a'), callback=self.parse) More examples and patterns From 3d77f74e4089c1a9700fb9e2bc62fc196176250c Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 5 Nov 2019 00:54:46 -0300 Subject: [PATCH 045/313] Download handlers: from_crawler factory method, take crawler instead of settings in __init__ --- scrapy/core/downloader/handlers/__init__.py | 8 ++- scrapy/core/downloader/handlers/datauri.py | 3 -- scrapy/core/downloader/handlers/file.py | 4 +- scrapy/core/downloader/handlers/ftp.py | 13 +++-- scrapy/core/downloader/handlers/http10.py | 20 +++++-- scrapy/core/downloader/handlers/http11.py | 12 +++-- scrapy/core/downloader/handlers/s3.py | 15 +++--- tests/test_downloader_handlers.py | 58 +++++++++++++-------- 8 files changed, 86 insertions(+), 47 deletions(-) diff --git a/scrapy/core/downloader/handlers/__init__.py b/scrapy/core/downloader/handlers/__init__.py index 0b55d32fa..e8beb2f5a 100644 --- a/scrapy/core/downloader/handlers/__init__.py +++ b/scrapy/core/downloader/handlers/__init__.py @@ -5,7 +5,7 @@ from twisted.internet import defer import six from scrapy.exceptions import NotSupported, NotConfigured from scrapy.utils.httpobj import urlparse_cached -from scrapy.utils.misc import load_object +from scrapy.utils.misc import create_instance, load_object from scrapy.utils.python import without_none_values from scrapy import signals @@ -48,7 +48,11 @@ class DownloadHandlers(object): dhcls = load_object(path) if skip_lazy and getattr(dhcls, 'lazy', True): return None - dh = dhcls(self._crawler.settings) + dh = create_instance( + dhcls, + self._crawler.settings, + self._crawler, + ) except NotConfigured as ex: self._notconfigured[scheme] = str(ex) return None diff --git a/scrapy/core/downloader/handlers/datauri.py b/scrapy/core/downloader/handlers/datauri.py index 9e5020753..97134e618 100644 --- a/scrapy/core/downloader/handlers/datauri.py +++ b/scrapy/core/downloader/handlers/datauri.py @@ -8,9 +8,6 @@ from scrapy.utils.decorators import defers class DataURIDownloadHandler(object): lazy = False - def __init__(self, settings): - super(DataURIDownloadHandler, self).__init__() - @defers def download_request(self, request, spider): uri = parse_data_uri(request.url) diff --git a/scrapy/core/downloader/handlers/file.py b/scrapy/core/downloader/handlers/file.py index 23f25d28d..d445ba2e1 100644 --- a/scrapy/core/downloader/handlers/file.py +++ b/scrapy/core/downloader/handlers/file.py @@ -1,4 +1,5 @@ from w3lib.url import file_uri_to_path + from scrapy.responsetypes import responsetypes from scrapy.utils.decorators import defers @@ -6,9 +7,6 @@ from scrapy.utils.decorators import defers class FileDownloadHandler(object): lazy = False - def __init__(self, settings): - pass - @defers def download_request(self, request, spider): filepath = file_uri_to_path(request.url) diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 39ed67a1a..7a98361ed 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -59,6 +59,7 @@ class ReceivedDataProtocol(Protocol): def close(self): self.body.close() if self.filename else self.body.seek(0) + _CODE_RE = re.compile(r"\d+") @@ -70,10 +71,14 @@ class FTPDownloadHandler(object): "default": 503, } - def __init__(self, settings): - self.default_user = settings['FTP_USER'] - self.default_password = settings['FTP_PASSWORD'] - self.passive_mode = settings['FTP_PASSIVE_MODE'] + def __init__(self, crawler): + self.default_user = crawler.settings['FTP_USER'] + self.default_password = crawler.settings['FTP_PASSWORD'] + self.passive_mode = crawler.settings['FTP_PASSIVE_MODE'] + + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) def download_request(self, request, spider): parsed_url = urlparse_cached(request) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index be7298531..ce0801bcc 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -1,6 +1,7 @@ """Download handlers for http and https schemes """ from twisted.internet import reactor + from scrapy.utils.misc import load_object, create_instance from scrapy.utils.python import to_unicode @@ -8,10 +9,15 @@ from scrapy.utils.python import to_unicode class HTTP10DownloadHandler(object): lazy = False - def __init__(self, settings): - self.HTTPClientFactory = load_object(settings['DOWNLOADER_HTTPCLIENTFACTORY']) - self.ClientContextFactory = load_object(settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) - self._settings = settings + def __init__(self, crawler): + self.HTTPClientFactory = load_object(crawler.settings['DOWNLOADER_HTTPCLIENTFACTORY']) + self.ClientContextFactory = load_object(crawler.settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) + self._crawler = crawler + self._settings = crawler.settings + + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) def download_request(self, request, spider): """Return a deferred for the HTTP download""" @@ -22,7 +28,11 @@ class HTTP10DownloadHandler(object): def _connect(self, factory): host, port = to_unicode(factory.host), factory.port if factory.scheme == b'https': - client_context_factory = create_instance(self.ClientContextFactory, settings=self._settings, crawler=None) + client_context_factory = create_instance( + self.ClientContextFactory, + settings=self._settings, + crawler=self._crawler, + ) return reactor.connectSSL(host, port, factory, client_context_factory) else: return reactor.connectTCP(host, port, factory) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 7d917cb74..b424a7999 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -30,7 +30,9 @@ logger = logging.getLogger(__name__) class HTTP11DownloadHandler(object): lazy = False - def __init__(self, settings): + def __init__(self, crawler): + settings = crawler.settings + self._pool = HTTPConnectionPool(reactor, persistent=True) self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN') self._pool._factory.noisy = False @@ -42,7 +44,7 @@ class HTTP11DownloadHandler(object): self._contextFactory = create_instance( self._contextFactoryClass, settings=settings, - crawler=None, + crawler=crawler, method=self._sslMethod, ) except TypeError: @@ -50,7 +52,7 @@ class HTTP11DownloadHandler(object): self._contextFactory = create_instance( self._contextFactoryClass, settings=settings, - crawler=None, + crawler=crawler, ) msg = """ '%s' does not accept `method` argument (type OpenSSL.SSL method,\ @@ -63,6 +65,10 @@ class HTTP11DownloadHandler(object): self._fail_on_dataloss = settings.getbool('DOWNLOAD_FAIL_ON_DATALOSS') self._disconnect_timeout = 1 + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) + def download_request(self, request, spider): """Return a deferred for the HTTP download""" agent = ScrapyAgent( diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index 808d1bf21..220296fb3 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -32,13 +32,12 @@ def _get_boto_connection(): class S3DownloadHandler(object): - def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None, \ - httpdownloadhandler=HTTPDownloadHandler, **kw): - + def __init__(self, crawler, aws_access_key_id=None, aws_secret_access_key=None, + httpdownloadhandler=HTTPDownloadHandler, **kw): if not aws_access_key_id: - aws_access_key_id = settings['AWS_ACCESS_KEY_ID'] + aws_access_key_id = crawler.settings['AWS_ACCESS_KEY_ID'] if not aws_secret_access_key: - aws_secret_access_key = settings['AWS_SECRET_ACCESS_KEY'] + aws_secret_access_key = crawler.settings['AWS_SECRET_ACCESS_KEY'] # If no credentials could be found anywhere, # consider this an anonymous connection request by default; @@ -67,7 +66,11 @@ class S3DownloadHandler(object): except Exception as ex: raise NotConfigured(str(ex)) - self._download_http = httpdownloadhandler(settings).download_request + self._download_http = httpdownloadhandler(crawler).download_request + + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) def download_request(self, request, spider): p = urlparse_cached(request) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 60124b93f..815342219 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -30,7 +30,6 @@ from scrapy.spiders import Spider from scrapy.http import Headers, Request from scrapy.http.response.text import TextResponse from scrapy.responsetypes import responsetypes -from scrapy.settings import Settings from scrapy.utils.test import get_crawler, skip_if_no_boto from scrapy.utils.python import to_bytes from scrapy.exceptions import NotConfigured @@ -45,6 +44,10 @@ class DummyDH(object): def __init__(self, crawler): pass + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) + class DummyLazyDH(object): # Default is lazy for backward compatibility @@ -52,6 +55,10 @@ class DummyLazyDH(object): def __init__(self, crawler): pass + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) + class OffDH(object): lazy = False @@ -59,6 +66,10 @@ class OffDH(object): def __init__(self, crawler): raise NotConfigured + @classmethod + def from_crawler(cls, crawler): + return cls(crawler) + class LoadTestCase(unittest.TestCase): @@ -106,7 +117,7 @@ class FileTestCase(unittest.TestCase): self.tmpname = self.mktemp() with open(self.tmpname + '^', 'w') as f: f.write('0123456789') - self.download_request = FileDownloadHandler(Settings()).download_request + self.download_request = FileDownloadHandler().download_request def tearDown(self): os.unlink(self.tmpname + '^') @@ -239,7 +250,7 @@ class HttpTestCase(unittest.TestCase): else: self.port = reactor.listenTCP(0, self.wrapper, interface=self.host) self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls(Settings()) + self.download_handler = self.download_handler_cls(get_crawler()) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -479,9 +490,9 @@ class Http11TestCase(HttpTestCase): return self.test_download_broken_content_allow_data_loss('broken-chunked') def test_download_broken_content_allow_data_loss_via_setting(self, url='broken'): - download_handler = self.download_handler_cls(Settings({ - 'DOWNLOAD_FAIL_ON_DATALOSS': False, - })) + download_handler = self.download_handler_cls( + get_crawler(settings_dict={'DOWNLOAD_FAIL_ON_DATALOSS': False}) + ) request = Request(self.getURL(url)) d = download_handler.download_request(request, Spider('foo')) d.addCallback(lambda r: r.flags) @@ -499,9 +510,9 @@ class Https11TestCase(Http11TestCase): @defer.inlineCallbacks def test_tls_logging(self): - download_handler = self.download_handler_cls(Settings({ - 'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True, - })) + download_handler = self.download_handler_cls( + get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True}) + ) try: with LogCapture() as log_capture: request = Request(self.getURL('file')) @@ -569,7 +580,8 @@ class Https11CustomCiphers(unittest.TestCase): interface=self.host) self.portno = self.port.getHost().port self.download_handler = self.download_handler_cls( - Settings({'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'})) + get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'}) + ) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -665,7 +677,7 @@ class HttpProxyTestCase(unittest.TestCase): wrapper = WrappingFactory(site) self.port = reactor.listenTCP(0, wrapper, interface='127.0.0.1') self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls(Settings()) + self.download_handler = self.download_handler_cls(get_crawler()) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -738,9 +750,10 @@ class S3AnonTestCase(unittest.TestCase): def setUp(self): skip_if_no_boto() - self.s3reqh = S3DownloadHandler(Settings(), - httpdownloadhandler=HttpDownloadHandlerMock, - #anon=True, # is implicit + self.s3reqh = S3DownloadHandler( + crawler=get_crawler(), + httpdownloadhandler=HttpDownloadHandlerMock, + #anon=True, # is implicit ) self.download_request = self.s3reqh.download_request self.spider = Spider('foo') @@ -766,9 +779,12 @@ class S3TestCase(unittest.TestCase): def setUp(self): skip_if_no_boto() - s3reqh = S3DownloadHandler(Settings(), self.AWS_ACCESS_KEY_ID, - self.AWS_SECRET_ACCESS_KEY, - httpdownloadhandler=HttpDownloadHandlerMock) + s3reqh = S3DownloadHandler( + get_crawler(), + self.AWS_ACCESS_KEY_ID, + self.AWS_SECRET_ACCESS_KEY, + httpdownloadhandler=HttpDownloadHandlerMock, + ) self.download_request = s3reqh.download_request self.spider = Spider('foo') @@ -788,7 +804,7 @@ class S3TestCase(unittest.TestCase): def test_extra_kw(self): try: - S3DownloadHandler(Settings(), extra_kw=True) + S3DownloadHandler(get_crawler(), extra_kw=True) except Exception as e: self.assertIsInstance(e, (TypeError, NotConfigured)) else: @@ -928,7 +944,7 @@ class BaseFTPTestCase(unittest.TestCase): self.factory = FTPFactory(portal=p) self.port = reactor.listenTCP(0, self.factory, interface="127.0.0.1") self.portNum = self.port.getHost().port - self.download_handler = FTPDownloadHandler(Settings()) + self.download_handler = FTPDownloadHandler(get_crawler()) self.addCleanup(self.port.stopListening) def tearDown(self): @@ -1042,7 +1058,7 @@ class AnonymousFTPTestCase(BaseFTPTestCase): userAnonymous=self.username) self.port = reactor.listenTCP(0, self.factory, interface="127.0.0.1") self.portNum = self.port.getHost().port - self.download_handler = FTPDownloadHandler(Settings()) + self.download_handler = FTPDownloadHandler(get_crawler()) self.addCleanup(self.port.stopListening) def tearDown(self): @@ -1052,7 +1068,7 @@ class AnonymousFTPTestCase(BaseFTPTestCase): class DataURITestCase(unittest.TestCase): def setUp(self): - self.download_handler = DataURIDownloadHandler(Settings()) + self.download_handler = DataURIDownloadHandler() self.download_request = self.download_handler.download_request self.spider = Spider('foo') From e43f37fff3cbdb5c4de7af21594c6062859a50e0 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 5 Nov 2019 16:18:42 -0300 Subject: [PATCH 046/313] Pass args/kwargs in S3DownloadHandler.from_crawler, update tests --- scrapy/core/downloader/handlers/s3.py | 4 ++-- tests/test_downloader_handlers.py | 22 +++++++++++----------- 2 files changed, 13 insertions(+), 13 deletions(-) diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index 220296fb3..99a3a7925 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -69,8 +69,8 @@ class S3DownloadHandler(object): self._download_http = httpdownloadhandler(crawler).download_request @classmethod - def from_crawler(cls, crawler): - return cls(crawler) + def from_crawler(cls, crawler, *args, **kwargs): + return cls(crawler, *args, **kwargs) def download_request(self, request, spider): p = urlparse_cached(request) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 815342219..eff5653f0 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -250,7 +250,7 @@ class HttpTestCase(unittest.TestCase): else: self.port = reactor.listenTCP(0, self.wrapper, interface=self.host) self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls(get_crawler()) + self.download_handler = self.download_handler_cls.from_crawler(get_crawler()) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -490,7 +490,7 @@ class Http11TestCase(HttpTestCase): return self.test_download_broken_content_allow_data_loss('broken-chunked') def test_download_broken_content_allow_data_loss_via_setting(self, url='broken'): - download_handler = self.download_handler_cls( + download_handler = self.download_handler_cls.from_crawler( get_crawler(settings_dict={'DOWNLOAD_FAIL_ON_DATALOSS': False}) ) request = Request(self.getURL(url)) @@ -510,7 +510,7 @@ class Https11TestCase(Http11TestCase): @defer.inlineCallbacks def test_tls_logging(self): - download_handler = self.download_handler_cls( + download_handler = self.download_handler_cls.from_crawler( get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True}) ) try: @@ -579,7 +579,7 @@ class Https11CustomCiphers(unittest.TestCase): 0, self.wrapper, ssl_context_factory(self.keyfile, self.certfile, cipher_string='CAMELLIA256-SHA'), interface=self.host) self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls( + self.download_handler = self.download_handler_cls.from_crawler( get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'}) ) self.download_request = self.download_handler.download_request @@ -677,7 +677,7 @@ class HttpProxyTestCase(unittest.TestCase): wrapper = WrappingFactory(site) self.port = reactor.listenTCP(0, wrapper, interface='127.0.0.1') self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls(get_crawler()) + self.download_handler = self.download_handler_cls.from_crawler(get_crawler()) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -750,7 +750,7 @@ class S3AnonTestCase(unittest.TestCase): def setUp(self): skip_if_no_boto() - self.s3reqh = S3DownloadHandler( + self.s3reqh = S3DownloadHandler.from_crawler( crawler=get_crawler(), httpdownloadhandler=HttpDownloadHandlerMock, #anon=True, # is implicit @@ -779,10 +779,10 @@ class S3TestCase(unittest.TestCase): def setUp(self): skip_if_no_boto() - s3reqh = S3DownloadHandler( - get_crawler(), - self.AWS_ACCESS_KEY_ID, - self.AWS_SECRET_ACCESS_KEY, + s3reqh = S3DownloadHandler.from_crawler( + crawler=get_crawler(), + aws_access_key_id=self.AWS_ACCESS_KEY_ID, + aws_secret_access_key=self.AWS_SECRET_ACCESS_KEY, httpdownloadhandler=HttpDownloadHandlerMock, ) self.download_request = s3reqh.download_request @@ -804,7 +804,7 @@ class S3TestCase(unittest.TestCase): def test_extra_kw(self): try: - S3DownloadHandler(get_crawler(), extra_kw=True) + S3DownloadHandler.from_crawler(get_crawler(), extra_kw=True) except Exception as e: self.assertIsInstance(e, (TypeError, NotConfigured)) else: From 2a9f5a0aefae83fc0b1dc161d117a265148ad1ef Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Tue, 3 Dec 2019 15:56:50 -0300 Subject: [PATCH 047/313] Skip invalid links when passing SelectorLists to Response.follow_all --- scrapy/http/response/text.py | 19 +++++++++++-------- tests/test_http_response.py | 12 ++++++++---- 2 files changed, 19 insertions(+), 12 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 6acf1026f..e3646b2d5 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -7,10 +7,10 @@ See documentation in docs/topics/request-response.rst from contextlib import suppress from typing import Generator +from urllib.parse import urljoin import parsel import six -from six.moves.urllib.parse import urljoin from w3lib.encoding import (html_body_declared_encoding, html_to_unicode, http_content_type_encoding, resolve_encoding) from w3lib.html import strip_html5_whitespace @@ -183,22 +183,25 @@ class TextResponse(Response): In addition, ``css`` and ``xpath`` arguments are accepted to perform the link extraction within the ``follow_all`` method (only one of ``urls``, ``css`` and ``xpath`` is accepted). - Note that when using the ``css`` or ``xpath`` parameters, this method will not produce - requests for selectors from which links cannot be obtained (for instance, anchor tags - without an ``href`` attribute) + Note that when passing a ``SelectorList`` as argument for the ``urls`` parameter or + using the ``css`` or ``xpath`` parameters, this method will not produce requests for + selectors from which links cannot be obtained (for instance, anchor tags without an + ``href`` attribute) """ arg_count = len(list(filter(None, (urls, css, xpath)))) if arg_count != 1: raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath') if not urls: if css: - selector_list = self.css(css) + urls = self.css(css) if xpath: - selector_list = self.xpath(xpath) + urls = self.xpath(xpath) + if isinstance(urls, parsel.SelectorList): + selectors = urls urls = [] - for selector in selector_list: + for sel in selectors: with suppress(_InvalidSelector): - urls.append(_url_from_selector(selector)) + urls.append(_url_from_selector(sel)) return super(TextResponse, self).follow_all( urls=urls, callback=callback, diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 36ccdfa1f..ce13650ce 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -579,8 +579,10 @@ class TextResponseTest(BaseResponseTest): 'http://example.com/page/4/', ] response = self._links_response_no_href() - extracted = [r.url for r in response.follow_all(css='.pagination a')] - self.assertEqual(expected, extracted) + extracted1 = [r.url for r in response.follow_all(css='.pagination a')] + self.assertEqual(expected, extracted1) + extracted2 = [r.url for r in response.follow_all(response.css('.pagination a'))] + self.assertEqual(expected, extracted2) def test_follow_all_xpath(self): expected = [ @@ -598,8 +600,10 @@ class TextResponseTest(BaseResponseTest): 'http://example.com/page/4/', ] response = self._links_response_no_href() - extracted = [r.url for r in response.follow_all(xpath='//div[@id="pagination"]/a')] - self.assertEqual(expected, extracted) + extracted1 = [r.url for r in response.follow_all(xpath='//div[@id="pagination"]/a')] + self.assertEqual(expected, extracted1) + extracted2 = [r.url for r in response.follow_all(response.xpath('//div[@id="pagination"]/a'))] + self.assertEqual(expected, extracted2) def test_follow_all_too_many_arguments(self): response = self._links_response() From 62778cf23f6e0cad5839f31f9da8b3c5778dcbdb Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 11 Sep 2019 15:40:44 -0300 Subject: [PATCH 048/313] Request: remove restriction about errback without callback --- scrapy/http/request/__init__.py | 1 - tests/test_http_request.py | 46 ++++++++++++++++++++++----------- 2 files changed, 31 insertions(+), 16 deletions(-) diff --git a/scrapy/http/request/__init__.py b/scrapy/http/request/__init__.py index 76a428199..61c0a4c9e 100644 --- a/scrapy/http/request/__init__.py +++ b/scrapy/http/request/__init__.py @@ -32,7 +32,6 @@ class Request(object_ref): raise TypeError('callback must be a callable, got %s' % type(callback).__name__) if errback is not None and not callable(errback): raise TypeError('errback must be a callable, got %s' % type(errback).__name__) - assert callback or not errback, "Cannot use errback without a callback" self.callback = callback self.errback = errback diff --git a/tests/test_http_request.py b/tests/test_http_request.py index 9df6ff67b..3449c7a40 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -246,25 +246,41 @@ class RequestTest(unittest.TestCase): self.assertRaises(AttributeError, setattr, r, 'url', 'http://example2.com') self.assertRaises(AttributeError, setattr, r, 'body', 'xxx') - def test_callback_is_callable(self): + def test_callback_and_errback(self): def a_function(): pass - r = self.request_class('http://example.com') - self.assertIsNone(r.callback) - r = self.request_class('http://example.com', a_function) - self.assertIs(r.callback, a_function) - with self.assertRaises(TypeError): - self.request_class('http://example.com', 'a_function') - def test_errback_is_callable(self): - def a_function(): - pass - r = self.request_class('http://example.com') - self.assertIsNone(r.errback) - r = self.request_class('http://example.com', a_function, errback=a_function) - self.assertIs(r.errback, a_function) + r1 = self.request_class('http://example.com') + self.assertIsNone(r1.callback) + self.assertIsNone(r1.errback) + + r2 = self.request_class('http://example.com', callback=a_function) + self.assertIs(r2.callback, a_function) + self.assertIsNone(r2.errback) + + r3 = self.request_class('http://example.com', errback=a_function) + self.assertIsNone(r3.callback) + self.assertIs(r3.errback, a_function) + + r4 = self.request_class( + url='http://example.com', + callback=a_function, + errback=a_function, + ) + self.assertIs(r4.callback, a_function) + self.assertIs(r4.errback, a_function) + + def test_callback_and_errback_type(self): with self.assertRaises(TypeError): - self.request_class('http://example.com', a_function, errback='a_function') + self.request_class('http://example.com', callback='a_function') + with self.assertRaises(TypeError): + self.request_class('http://example.com', errback='a_function') + with self.assertRaises(TypeError): + self.request_class( + url='http://example.com', + callback='a_function', + errback='a_function', + ) def test_from_curl(self): # Note: more curated tests regarding curl conversion are in From 1b35260625c3ffec9885265d9ac92771ade67ad9 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 25 Jul 2019 18:18:34 +0500 Subject: [PATCH 049/313] Add a test for downloader middlewares using Deferreds. --- tests/test_downloadermiddleware.py | 29 +++++++++++++++++++++++++++++ 1 file changed, 29 insertions(+) diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 6b9a5bee8..1b81ea949 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -1,5 +1,6 @@ from unittest import mock +from twisted.internet.defer import Deferred from twisted.trial.unittest import TestCase from twisted.python.failure import Failure @@ -177,3 +178,31 @@ class ProcessExceptionInvalidOutput(ManagerTestCase): dfd.addBoth(results.append) self.assertIsInstance(results[0], Failure) self.assertIsInstance(results[0].value, _InvalidOutput) + + +class MiddlewareUsingDeferreds(ManagerTestCase): + """Middlewares using Deferreds should work""" + + def test_deferred(self): + resp = Response('http://example.com/index.html') + + class DeferredMiddleware: + def cb(self, result): + return result + + def process_request(self, request, spider): + d = Deferred() + d.addCallback(self.cb) + d.callback(resp) + return d + + self.mwman._add_middleware(DeferredMiddleware()) + req = Request('http://example.com/index.html') + download_func = mock.MagicMock() + dfd = self.mwman.download(download_func, req, self.spider) + results = [] + dfd.addBoth(results.append) + self._wait(dfd) + + self.assertIs(results[0], resp) + self.assertFalse(download_func.called) From 1b437bbe9fa0eb9736f35e510d486805706c783e Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 30 Jul 2019 19:02:16 +0500 Subject: [PATCH 050/313] Install the asyncio reactor on "import scrapy". --- scrapy/__init__.py | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 230e5cee3..41eaee959 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -23,6 +23,28 @@ import warnings warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted') del warnings +# Install twisted asyncio loop +def _install_asyncio_reactor(): + global asyncio_supported + try: + import asyncio + from twisted.internet import asyncioreactor + except ImportError: + pass + else: + from twisted.internet.error import ReactorAlreadyInstalledError + try: + asyncioreactor.install(asyncio.get_event_loop()) + asyncio_supported = True + except ReactorAlreadyInstalledError: + import twisted.internet.reactor + if isinstance(twisted.internet.reactor, + asyncioreactor.AsyncioSelectorReactor): + asyncio_supported = True +asyncio_supported = False +_install_asyncio_reactor() +del _install_asyncio_reactor + # Apply monkey patches to fix issues in external libraries from . import _monkeypatches del _monkeypatches From 9777639533373951652f3865a6321a4dff73246a Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 30 Jul 2019 19:02:59 +0500 Subject: [PATCH 051/313] Run tests using the asyncio reactor. --- pytest.ini | 1 + tests/mockserver.py | 3 +++ tests/requirements-py3.txt | 3 ++- 3 files changed, 6 insertions(+), 1 deletion(-) diff --git a/pytest.ini b/pytest.ini index 33c34b8e8..6c4c21baf 100644 --- a/pytest.ini +++ b/pytest.ini @@ -5,6 +5,7 @@ python_classes= addopts = --assert=plain --doctest-modules + --reactor=asyncio --ignore=docs/_ext --ignore=docs/conf.py --ignore=docs/news.rst diff --git a/tests/mockserver.py b/tests/mockserver.py index 7ebb8bb62..b6aee009a 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -6,6 +6,9 @@ from subprocess import Popen, PIPE from OpenSSL import SSL from six.moves.urllib.parse import urlencode + +import scrapy # needed before importing twisted.internet.reactor + from twisted.web.server import Site, NOT_DONE_YET from twisted.web.resource import Resource from twisted.web.static import File diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index c4bc1f278..26ab08b04 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -4,7 +4,8 @@ mitmproxy; python_version >= '3.6' mitmproxy==3.0.4; python_version < '3.6' pytest pytest-cov -pytest-twisted +#pytest-twisted +-e git+https://github.com/pytest-dev/pytest-twisted@81b91f17#egg=pytest-twisted pytest-xdist sybil testfixtures From 63c3c62305a8c9c52d02ff5524301b7d45eb5724 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Tue, 30 Jul 2019 19:45:56 +0500 Subject: [PATCH 052/313] Add utils.deferred_from_coro. --- scrapy/utils/defer.py | 22 ++++++++++++++++++++++ 1 file changed, 22 insertions(+) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index c5916c21c..1f6a2584c 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -1,10 +1,14 @@ """ Helper functions for dealing with Twisted deferreds """ +import asyncio +import asyncio.futures +import inspect from twisted.internet import defer, reactor, task from twisted.python import failure +from scrapy import asyncio_supported from scrapy.exceptions import IgnoreRequest @@ -113,3 +117,21 @@ def iter_errback(iterable, errback, *a, **kw): break except Exception: errback(failure.Failure(), *a, **kw) + + +def isfuture(o): + # workaround for Python before 3.5.3 not having asyncio.isfuture + if hasattr(asyncio, 'isfuture'): + return asyncio.isfuture(o) + return isinstance(o, asyncio.futures.Future) + + +def deferred_from_coro(o): + """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" + if isinstance(o, defer.Deferred): + return o + if asyncio.iscoroutine(o) or isfuture(o) or inspect.isawaitable(o): + if not asyncio_supported: + raise TypeError('Using coroutines requires installing AsyncioSelectorReactor') + return defer.Deferred.fromFuture(asyncio.ensure_future(o)) + return o From 8d8fbddbde133a94bb8741e48fabec05437a3df9 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 21 Aug 2019 00:07:08 +0500 Subject: [PATCH 053/313] Switch to the released version of pytest-twisted. --- tests/requirements-py3.txt | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index 26ab08b04..2ac434f41 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -4,8 +4,7 @@ mitmproxy; python_version >= '3.6' mitmproxy==3.0.4; python_version < '3.6' pytest pytest-cov -#pytest-twisted --e git+https://github.com/pytest-dev/pytest-twisted@81b91f17#egg=pytest-twisted +pytest-twisted >= 1.11 pytest-xdist sybil testfixtures From b04b541372b219a02bb5fda2cc15cfd9fa1aac66 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 21 Aug 2019 17:14:46 +0500 Subject: [PATCH 054/313] Install the asyncio reactor only in scrapy.cmdline. --- scrapy/__init__.py | 22 ---------------------- scrapy/cmdline.py | 8 +++++++- scrapy/utils/asyncio.py | 26 ++++++++++++++++++++++++++ scrapy/utils/defer.py | 5 +++-- tests/mockserver.py | 2 -- 5 files changed, 36 insertions(+), 27 deletions(-) create mode 100644 scrapy/utils/asyncio.py diff --git a/scrapy/__init__.py b/scrapy/__init__.py index 41eaee959..230e5cee3 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -23,28 +23,6 @@ import warnings warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted') del warnings -# Install twisted asyncio loop -def _install_asyncio_reactor(): - global asyncio_supported - try: - import asyncio - from twisted.internet import asyncioreactor - except ImportError: - pass - else: - from twisted.internet.error import ReactorAlreadyInstalledError - try: - asyncioreactor.install(asyncio.get_event_loop()) - asyncio_supported = True - except ReactorAlreadyInstalledError: - import twisted.internet.reactor - if isinstance(twisted.internet.reactor, - asyncioreactor.AsyncioSelectorReactor): - asyncio_supported = True -asyncio_supported = False -_install_asyncio_reactor() -del _install_asyncio_reactor - # Apply monkey patches to fix issues in external libraries from . import _monkeypatches del _monkeypatches diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index 418dc1ac9..d66f0cc2d 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -7,9 +7,9 @@ import inspect import pkg_resources import scrapy -from scrapy.crawler import CrawlerProcess from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError +from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_supported from scrapy.utils.misc import walk_modules from scrapy.utils.project import inside_project, get_project_settings from scrapy.utils.python import garbage_collect @@ -121,6 +121,10 @@ def execute(argv=None, settings=None): settings['EDITOR'] = editor check_deprecated_settings(settings) + # needs to be before _get_commands_dict() as that imports the command modules + # which may import twisted.internet.reactor + install_asyncio_reactor() + inproject = inside_project() cmds = _get_commands_dict(settings, inproject) cmdname = _pop_command_name(argv) @@ -142,6 +146,8 @@ def execute(argv=None, settings=None): opts, args = parser.parse_args(args=argv[1:]) _run_print_help(parser, cmd.process_options, args, opts) + # needs to be after install_asyncio_reactor() as it imports twisted.internet.reactor + from scrapy.crawler import CrawlerProcess cmd.crawler_process = CrawlerProcess(settings) _run_print_help(parser, _run_command, cmd, args, opts) sys.exit(cmd.exitcode) diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py new file mode 100644 index 000000000..e9e3bdd88 --- /dev/null +++ b/scrapy/utils/asyncio.py @@ -0,0 +1,26 @@ +#coding: utf-8 + + +def install_asyncio_reactor(): + """ Tries to install AsyncioSelectorReactor + """ + try: + import asyncio + from twisted.internet import asyncioreactor + except ImportError: + pass + else: + from twisted.internet.error import ReactorAlreadyInstalledError + try: + asyncioreactor.install(asyncio.get_event_loop()) + except ReactorAlreadyInstalledError: + pass + + +def is_asyncio_supported(): + try: + import twisted.internet.reactor + from twisted.internet import asyncioreactor + return isinstance(twisted.internet.reactor, asyncioreactor.AsyncioSelectorReactor) + except ImportError: + return False diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 1f6a2584c..955fc820a 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -8,8 +8,9 @@ import inspect from twisted.internet import defer, reactor, task from twisted.python import failure -from scrapy import asyncio_supported from scrapy.exceptions import IgnoreRequest +from scrapy.utils.asyncio import is_asyncio_supported + def defer_fail(_failure): @@ -131,7 +132,7 @@ def deferred_from_coro(o): if isinstance(o, defer.Deferred): return o if asyncio.iscoroutine(o) or isfuture(o) or inspect.isawaitable(o): - if not asyncio_supported: + if not is_asyncio_supported(): raise TypeError('Using coroutines requires installing AsyncioSelectorReactor') return defer.Deferred.fromFuture(asyncio.ensure_future(o)) return o diff --git a/tests/mockserver.py b/tests/mockserver.py index b6aee009a..d09fbc171 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -7,8 +7,6 @@ from subprocess import Popen, PIPE from OpenSSL import SSL from six.moves.urllib.parse import urlencode -import scrapy # needed before importing twisted.internet.reactor - from twisted.web.server import Site, NOT_DONE_YET from twisted.web.resource import Resource from twisted.web.static import File From 2fbe7d49dc084b7770cc4dc6bbe65eb380b5f498 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 21 Aug 2019 17:16:33 +0500 Subject: [PATCH 055/313] Log asyncio support on spider start. --- scrapy/utils/log.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index e07fb8698..b74b7a4af 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -11,6 +11,7 @@ from twisted.python import log as twisted_log import scrapy from scrapy.settings import Settings from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.asyncio import is_asyncio_supported from scrapy.utils.versions import scrapy_components_versions @@ -148,6 +149,8 @@ def log_scrapy_info(settings): {'versions': ", ".join("%s %s" % (name, version) for name, version in scrapy_components_versions() if name != "Scrapy")}) + if is_asyncio_supported(): + logger.debug("Asyncio support enabled") class StreamLogger(object): From cc19ab5439f20ba6995528542cc064ddab86273c Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 22 Aug 2019 18:15:02 +0500 Subject: [PATCH 056/313] Add tests that check asyncio support. --- conftest.py | 4 ++++ tests/test_commands.py | 7 +++++++ tests/test_crawler.py | 20 +++++++++++++++++++- tests/test_utils_asyncio.py | 17 +++++++++++++++++ tox.ini | 6 ++++++ 5 files changed, 53 insertions(+), 1 deletion(-) create mode 100644 tests/test_utils_asyncio.py diff --git a/conftest.py b/conftest.py index d54ce155c..24e31f130 100644 --- a/conftest.py +++ b/conftest.py @@ -27,3 +27,7 @@ def pytest_collection_modifyitems(session, config, items): items[:] = [item for item in items if isinstance(item, Flake8Item)] except ImportError: pass + +@pytest.fixture() +def reactor_pytest(request): + request.cls.reactor_pytest = request.config.getoption("--reactor") diff --git a/tests/test_commands.py b/tests/test_commands.py index 536379170..8aa7ee109 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -9,6 +9,7 @@ from tempfile import mkdtemp from contextlib import contextmanager from threading import Timer +from pytest import mark from twisted.trial import unittest import scrapy @@ -178,6 +179,7 @@ class MiscCommandsTest(CommandTest): self.assertEqual(0, self.call('list')) +@mark.usefixtures('reactor_pytest') class RunSpiderCommandTest(CommandTest): debug_log_spider = """ @@ -295,6 +297,11 @@ class BadSpider(scrapy.Spider): self.assertIn("start_requests", log) self.assertIn("badspider.py", log) + def test_asyncio_supported(self): + if self.reactor_pytest == 'asyncio': + log = self.get_log(self.debug_log_spider) + self.assertIn("DEBUG: Asyncio support enabled", log) + class BenchCommandTest(CommandTest): diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 8eb2389e2..151acb459 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -1,14 +1,16 @@ import logging import warnings +from pytest import raises, mark +from testfixtures import LogCapture from twisted.internet import defer from twisted.trial import unittest -from pytest import raises import scrapy from scrapy.crawler import Crawler, CrawlerRunner, CrawlerProcess from scrapy.settings import Settings, default_settings from scrapy.spiderloader import SpiderLoader +from scrapy.utils.asyncio import is_asyncio_supported from scrapy.utils.log import configure_logging, get_scrapy_root_handler from scrapy.utils.spider import DefaultSpider from scrapy.utils.misc import load_object @@ -203,6 +205,15 @@ class NoRequestsSpider(scrapy.Spider): return [] +class AsyncioSpider(scrapy.Spider): + name = 'asyncio' + + def start_requests(self): + self.logger.info('Asyncio support: %s', is_asyncio_supported()) + return [] + + +@mark.usefixtures('reactor_pytest') class CrawlerRunnerHasSpider(unittest.TestCase): @defer.inlineCallbacks @@ -245,3 +256,10 @@ class CrawlerRunnerHasSpider(unittest.TestCase): yield runner.crawl(NoRequestsSpider) self.assertEqual(runner.bootstrap_failed, True) + + @defer.inlineCallbacks + def test_asyncio_supported(self): + runner = CrawlerRunner() + with LogCapture() as log: + yield runner.crawl(AsyncioSpider) + log.check_present(('asyncio', 'INFO', 'Asyncio support: %s' % (self.reactor_pytest == 'asyncio'))) diff --git a/tests/test_utils_asyncio.py b/tests/test_utils_asyncio.py new file mode 100644 index 000000000..e34d3002a --- /dev/null +++ b/tests/test_utils_asyncio.py @@ -0,0 +1,17 @@ +from unittest import TestCase + +from pytest import mark + +from scrapy.utils.asyncio import is_asyncio_supported, install_asyncio_reactor + + +@mark.usefixtures('reactor_pytest') +class AsyncioTest(TestCase): + + def test_is_asyncio_supported(self): + # the result should depend only on the pytest --reactor argument + self.assertEquals(is_asyncio_supported(), self.reactor_pytest == 'asyncio') + + def test_install_asyncio_reactor(self): + # this should do nothing + install_asyncio_reactor() diff --git a/tox.ini b/tox.ini index fd75d18e2..844956e5f 100644 --- a/tox.ini +++ b/tox.ini @@ -106,3 +106,9 @@ deps = {[testenv]deps} reppy robotexclusionrulesparser + +[testenv:py38-no-asyncio] +basepython = python3.8 +deps = {[testenv]deps} +commands = + py.test --cov=scrapy --cov-report= --reactor=default {posargs:scrapy tests} From f41c2f3874d2f9deac365d633a71f032e1339e3c Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 22 Aug 2019 21:24:30 +0500 Subject: [PATCH 057/313] Add py38-no-asyncio to Travis. --- .travis.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.travis.yml b/.travis.yml index 9f477e860..fdf40fdf1 100644 --- a/.travis.yml +++ b/.travis.yml @@ -25,6 +25,8 @@ matrix: python: 3.8 - env: TOXENV=py38-extra-deps python: 3.8 + - env: TOXENV=py38-no-asyncio + python: 3.8 - env: TOXENV=docs python: 3.6 install: From 3ba25ccbd3b456024b1d350645407557c81d73c7 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Fri, 8 Nov 2019 00:09:28 +0500 Subject: [PATCH 058/313] Don't use asyncio.iscoroutine, as it is True for generators. --- scrapy/utils/defer.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 955fc820a..30163d2fb 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -131,7 +131,7 @@ def deferred_from_coro(o): """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" if isinstance(o, defer.Deferred): return o - if asyncio.iscoroutine(o) or isfuture(o) or inspect.isawaitable(o): + if isfuture(o) or inspect.isawaitable(o): if not is_asyncio_supported(): raise TypeError('Using coroutines requires installing AsyncioSelectorReactor') return defer.Deferred.fromFuture(asyncio.ensure_future(o)) From 794cf71806a94ba68238a93e75c396f355159ab5 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 14 Nov 2019 13:27:21 +0500 Subject: [PATCH 059/313] Fix or ignore flake8 problems. --- pytest.ini | 2 ++ scrapy/cmdline.py | 2 +- scrapy/utils/asyncio.py | 3 --- 3 files changed, 3 insertions(+), 4 deletions(-) diff --git a/pytest.ini b/pytest.ini index 6c4c21baf..8b97237c3 100644 --- a/pytest.ini +++ b/pytest.ini @@ -116,6 +116,7 @@ flake8-ignore = scrapy/spiders/feed.py E501 E261 scrapy/spiders/sitemap.py E501 # scrapy/utils + scrapy/utils/asyncio.py E501 scrapy/utils/benchserver.py E501 scrapy/utils/conf.py E402 E502 E501 scrapy/utils/console.py E261 E306 E305 @@ -227,6 +228,7 @@ flake8-ignore = tests/test_spidermiddleware_output_chain.py E501 W293 E226 tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E261 E124 E501 E241 E121 tests/test_squeues.py E501 E701 E741 + tests/test_utils_asyncio.py E501 tests/test_utils_conf.py E501 E303 E128 tests/test_utils_curl.py E501 tests/test_utils_datatypes.py E402 E501 E305 diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index d66f0cc2d..213e99bc0 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -9,7 +9,7 @@ import pkg_resources import scrapy from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError -from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_supported +from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.misc import walk_modules from scrapy.utils.project import inside_project, get_project_settings from scrapy.utils.python import garbage_collect diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py index e9e3bdd88..f732774f1 100644 --- a/scrapy/utils/asyncio.py +++ b/scrapy/utils/asyncio.py @@ -1,6 +1,3 @@ -#coding: utf-8 - - def install_asyncio_reactor(): """ Tries to install AsyncioSelectorReactor """ From c079d5002bae90dbd85bee1f61fdc359b9f39d29 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Thu, 21 Nov 2019 23:40:16 +0500 Subject: [PATCH 060/313] Run tests without asyncio support by default, add py35-asyncio and py38-asyncio envs. --- pytest.ini | 1 - tox.ini | 10 ++++++++-- 2 files changed, 8 insertions(+), 3 deletions(-) diff --git a/pytest.ini b/pytest.ini index 8b97237c3..336ef041d 100644 --- a/pytest.ini +++ b/pytest.ini @@ -5,7 +5,6 @@ python_classes= addopts = --assert=plain --doctest-modules - --reactor=asyncio --ignore=docs/_ext --ignore=docs/conf.py --ignore=docs/news.rst diff --git a/tox.ini b/tox.ini index 844956e5f..a4edae439 100644 --- a/tox.ini +++ b/tox.ini @@ -107,8 +107,14 @@ deps = reppy robotexclusionrulesparser -[testenv:py38-no-asyncio] +[testenv:py35-asyncio] +basepython = python3.5 +deps = {[testenv]deps} +commands = + py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests} + +[testenv:py38-asyncio] basepython = python3.8 deps = {[testenv]deps} commands = - py.test --cov=scrapy --cov-report= --reactor=default {posargs:scrapy tests} + py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests} From ed34ce14c0c06d4539d4fdeb0ad014f4e6fb5b94 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 4 Dec 2019 21:32:16 +0500 Subject: [PATCH 061/313] Add the ASYNCIO_SUPPORT setting, reshuffle other logic accordingly. --- docs/topics/settings.rst | 25 +++++++++++++++++++++++++ scrapy/cmdline.py | 8 ++------ scrapy/commands/crawl.py | 3 +++ scrapy/commands/runspider.py | 3 +++ scrapy/settings/default_settings.py | 2 ++ scrapy/utils/asyncio.py | 2 +- scrapy/utils/defer.py | 17 +++++++++++------ scrapy/utils/log.py | 11 ++++++++--- tests/test_commands.py | 13 +++++++------ tests/test_crawler.py | 24 +++++++++++++++++++++--- tests/test_utils_asyncio.py | 6 +++--- 11 files changed, 86 insertions(+), 28 deletions(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index a1d15a760..43f59f7cc 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -160,6 +160,31 @@ to any particular component. In that case the module of that component will be shown, typically an extension, middleware or pipeline. It also means that the component must be enabled in order for the setting to have any effect. +.. setting:: ASYNCIO_SUPPORT + +ASYNCIO_SUPPORT +--------------- + +Default: ``False`` + +Whether to support ``async def`` methods and callbacks which use code that +requires an asyncio loop. + +If an ``async def`` coroutine doesn't require the asyncio loop, it will work +even if this is set to ``False``. Coroutines that require the asyncio loop may +silently fail to run or raise errors unless this is set to ``True``. + +When this option is set to ``True``, Scrapy will require +:class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`. It will +install this reactor if no reactor is installed yet, such as when using the +``scrapy`` script or :class:`~scrapy.crawler.CrawlerProcess`. If you are using +:class:`~scrapy.crawler.CrawlerRunner`, you need to install the correct reactor +manually. + +The default value for this option is currently ``False`` to maintain backward +compatibility and avoid possible problems caused by using a different Twisted +reactor. + .. setting:: AWS_ACCESS_KEY_ID AWS_ACCESS_KEY_ID diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index 213e99bc0..ce030cf75 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -9,7 +9,6 @@ import pkg_resources import scrapy from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError -from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.misc import walk_modules from scrapy.utils.project import inside_project, get_project_settings from scrapy.utils.python import garbage_collect @@ -121,10 +120,6 @@ def execute(argv=None, settings=None): settings['EDITOR'] = editor check_deprecated_settings(settings) - # needs to be before _get_commands_dict() as that imports the command modules - # which may import twisted.internet.reactor - install_asyncio_reactor() - inproject = inside_project() cmds = _get_commands_dict(settings, inproject) cmdname = _pop_command_name(argv) @@ -146,7 +141,8 @@ def execute(argv=None, settings=None): opts, args = parser.parse_args(args=argv[1:]) _run_print_help(parser, cmd.process_options, args, opts) - # needs to be after install_asyncio_reactor() as it imports twisted.internet.reactor + # needs to be after cmd.process_options() as it imports twisted.internet.reactor + # while commands may want to install the asyncio reactor from scrapy.crawler import CrawlerProcess cmd.crawler_process = CrawlerProcess(settings) _run_print_help(parser, _run_command, cmd, args, opts) diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index 8093fd402..e2e69be49 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -1,5 +1,6 @@ import os from scrapy.commands import ScrapyCommand +from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.conf import arglist_to_dict from scrapy.utils.python import without_none_values from scrapy.exceptions import UsageError @@ -26,6 +27,8 @@ class Command(ScrapyCommand): def process_options(self, args, opts): ScrapyCommand.process_options(self, args, opts) + if self.settings.getbool('ASYNCIO_SUPPORT'): + install_asyncio_reactor() try: opts.spargs = arglist_to_dict(opts.spargs) except ValueError: diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index 57d8471ca..ebd4eb620 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -2,6 +2,7 @@ import sys import os from importlib import import_module +from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.spider import iter_spider_classes from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError @@ -50,6 +51,8 @@ class Command(ScrapyCommand): def process_options(self, args, opts): ScrapyCommand.process_options(self, args, opts) + if self.settings.getbool('ASYNCIO_SUPPORT'): + install_asyncio_reactor() try: opts.spargs = arglist_to_dict(opts.spargs) except ValueError: diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index 5c9678c01..c9097bd1f 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -19,6 +19,8 @@ from os.path import join, abspath, dirname AJAXCRAWL_ENABLED = False +ASYNCIO_SUPPORT = False + AUTOTHROTTLE_ENABLED = False AUTOTHROTTLE_DEBUG = False AUTOTHROTTLE_MAX_DELAY = 60.0 diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py index f732774f1..b5d5f92d9 100644 --- a/scrapy/utils/asyncio.py +++ b/scrapy/utils/asyncio.py @@ -14,7 +14,7 @@ def install_asyncio_reactor(): pass -def is_asyncio_supported(): +def is_asyncio_reactor_installed(): try: import twisted.internet.reactor from twisted.internet import asyncioreactor diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 30163d2fb..3b7ef75ab 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -9,8 +9,7 @@ from twisted.internet import defer, reactor, task from twisted.python import failure from scrapy.exceptions import IgnoreRequest -from scrapy.utils.asyncio import is_asyncio_supported - +from scrapy.utils.asyncio import is_asyncio_reactor_installed def defer_fail(_failure): @@ -127,12 +126,18 @@ def isfuture(o): return isinstance(o, asyncio.futures.Future) -def deferred_from_coro(o): +def deferred_from_coro(o, asyncio_enabled=False): """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" if isinstance(o, defer.Deferred): return o if isfuture(o) or inspect.isawaitable(o): - if not is_asyncio_supported(): - raise TypeError('Using coroutines requires installing AsyncioSelectorReactor') - return defer.Deferred.fromFuture(asyncio.ensure_future(o)) + if not asyncio_enabled: + # wrapping the coroutine directly into a Deferred, this doesn't work correctly with coroutines + # that use asyncio, e.g. "await asyncio.sleep(1)" + return defer.ensureDeferred(o) + else: + # wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor + if not is_asyncio_reactor_installed(): + raise TypeError('Using coroutines requires installing AsyncioSelectorReactor') + return defer.Deferred.fromFuture(asyncio.ensure_future(o)) return o diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index b74b7a4af..8c56cfa42 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -11,7 +11,7 @@ from twisted.python import log as twisted_log import scrapy from scrapy.settings import Settings from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.asyncio import is_asyncio_supported +from scrapy.utils.asyncio import is_asyncio_reactor_installed from scrapy.utils.versions import scrapy_components_versions @@ -149,8 +149,13 @@ def log_scrapy_info(settings): {'versions': ", ".join("%s %s" % (name, version) for name, version in scrapy_components_versions() if name != "Scrapy")}) - if is_asyncio_supported(): - logger.debug("Asyncio support enabled") + if settings.getbool('ASYNCIO_SUPPORT'): + if is_asyncio_reactor_installed(): + logger.debug("Asyncio support enabled") + else: + logger.error("ASYNCIO_SUPPORT is on but the Twisted asyncio " + "reactor is not installed, this is not supported " + "and asyncio coroutines will not work.") class StreamLogger(object): diff --git a/tests/test_commands.py b/tests/test_commands.py index 8aa7ee109..3b64bfa23 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -9,7 +9,6 @@ from tempfile import mkdtemp from contextlib import contextmanager from threading import Timer -from pytest import mark from twisted.trial import unittest import scrapy @@ -179,7 +178,6 @@ class MiscCommandsTest(CommandTest): self.assertEqual(0, self.call('list')) -@mark.usefixtures('reactor_pytest') class RunSpiderCommandTest(CommandTest): debug_log_spider = """ @@ -297,10 +295,13 @@ class BadSpider(scrapy.Spider): self.assertIn("start_requests", log) self.assertIn("badspider.py", log) - def test_asyncio_supported(self): - if self.reactor_pytest == 'asyncio': - log = self.get_log(self.debug_log_spider) - self.assertIn("DEBUG: Asyncio support enabled", log) + def test_asyncio_support_true(self): + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_SUPPORT=True']) + self.assertIn("DEBUG: Asyncio support enabled", log) + + def test_asyncio_support_false(self): + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_SUPPORT=False']) + self.assertNotIn("DEBUG: Asyncio support enabled", log) class BenchCommandTest(CommandTest): diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 151acb459..3ac45ca1d 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -10,7 +10,7 @@ import scrapy from scrapy.crawler import Crawler, CrawlerRunner, CrawlerProcess from scrapy.settings import Settings, default_settings from scrapy.spiderloader import SpiderLoader -from scrapy.utils.asyncio import is_asyncio_supported +from scrapy.utils.asyncio import is_asyncio_reactor_installed from scrapy.utils.log import configure_logging, get_scrapy_root_handler from scrapy.utils.spider import DefaultSpider from scrapy.utils.misc import load_object @@ -209,7 +209,7 @@ class AsyncioSpider(scrapy.Spider): name = 'asyncio' def start_requests(self): - self.logger.info('Asyncio support: %s', is_asyncio_supported()) + self.logger.info('Asyncio support: %s', is_asyncio_reactor_installed()) return [] @@ -258,7 +258,25 @@ class CrawlerRunnerHasSpider(unittest.TestCase): self.assertEqual(runner.bootstrap_failed, True) @defer.inlineCallbacks - def test_asyncio_supported(self): + def test_crawler_process_asyncio_supported_true(self): + with LogCapture(level=logging.DEBUG) as log: + runner = CrawlerProcess(settings={'ASYNCIO_SUPPORT': True}) + yield runner.crawl(NoRequestsSpider) + if self.reactor_pytest == 'asyncio': + self.assertIn("Asyncio support enabled", str(log)) + else: + self.assertNotIn("Asyncio support enabled", str(log)) + self.assertIn("ASYNCIO_SUPPORT is on but the Twisted asyncio reactor is not installed", str(log)) + + @defer.inlineCallbacks + def test_crawler_process_asyncio_supported_false(self): + runner = CrawlerProcess(settings={'ASYNCIO_SUPPORT': False}) + with LogCapture(level=logging.DEBUG) as log: + yield runner.crawl(NoRequestsSpider) + self.assertNotIn("Asyncio support enabled", str(log)) + + @defer.inlineCallbacks + def test_crawler_runner_asyncio_supported(self): runner = CrawlerRunner() with LogCapture() as log: yield runner.crawl(AsyncioSpider) diff --git a/tests/test_utils_asyncio.py b/tests/test_utils_asyncio.py index e34d3002a..a6ba24876 100644 --- a/tests/test_utils_asyncio.py +++ b/tests/test_utils_asyncio.py @@ -2,15 +2,15 @@ from unittest import TestCase from pytest import mark -from scrapy.utils.asyncio import is_asyncio_supported, install_asyncio_reactor +from scrapy.utils.asyncio import is_asyncio_reactor_installed, install_asyncio_reactor @mark.usefixtures('reactor_pytest') class AsyncioTest(TestCase): - def test_is_asyncio_supported(self): + def test_is_asyncio_reactor_installed(self): # the result should depend only on the pytest --reactor argument - self.assertEquals(is_asyncio_supported(), self.reactor_pytest == 'asyncio') + self.assertEquals(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio') def test_install_asyncio_reactor(self): # this should do nothing From 97fb61cec846641eb1c8e224ae24e55558746f4f Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 4 Dec 2019 21:53:07 +0500 Subject: [PATCH 062/313] Move an import to postpone another "import twisted.internet.reactor". --- scrapy/commands/shell.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/scrapy/commands/shell.py b/scrapy/commands/shell.py index e05084272..7516e2aba 100644 --- a/scrapy/commands/shell.py +++ b/scrapy/commands/shell.py @@ -6,7 +6,6 @@ See documentation in docs/topics/shell.rst from threading import Thread from scrapy.commands import ScrapyCommand -from scrapy.shell import Shell from scrapy.http import Request from scrapy.utils.spider import spidercls_for_request, DefaultSpider from scrapy.utils.url import guess_scheme @@ -70,6 +69,8 @@ class Command(ScrapyCommand): self._start_crawler_thread() + # moved from the top-level because it imports twisted.internet.reactor + from scrapy.shell import Shell shell = Shell(crawler, update_vars=self.update_vars, code=opts.code) shell.start(url=url, redirect=not opts.no_redirect) From 0b9f29215ff7f81203efbbc25b8e9cf3e9719920 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin Date: Wed, 4 Dec 2019 22:06:35 +0500 Subject: [PATCH 063/313] Update .travis.yml. --- .travis.yml | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/.travis.yml b/.travis.yml index fdf40fdf1..98dab01f2 100644 --- a/.travis.yml +++ b/.travis.yml @@ -17,6 +17,8 @@ matrix: python: 3.5 - env: TOXENV=py35-pinned python: 3.5 + - env: TOXENV=py35-asyncio + python: 3.5 - env: TOXENV=py36 python: 3.6 - env: TOXENV=py37 @@ -25,7 +27,7 @@ matrix: python: 3.8 - env: TOXENV=py38-extra-deps python: 3.8 - - env: TOXENV=py38-no-asyncio + - env: TOXENV=py38-asyncio python: 3.8 - env: TOXENV=docs python: 3.6 From 83b8046fdcae660276634a03b99d51c629fa8b7c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Fri, 8 Nov 2019 16:26:46 +0100 Subject: [PATCH 064/313] Do not indent doctests from the documentation unnecessarily --- docs/intro/tutorial.rst | 120 +++---- docs/news.rst | 12 +- docs/topics/developer-tools.rst | 12 +- docs/topics/dynamic-content.rst | 28 +- docs/topics/items.rst | 106 +++--- docs/topics/leaks.rst | 140 ++++---- docs/topics/loaders.rst | 114 ++++--- docs/topics/selectors.rst | 565 ++++++++++++++++---------------- docs/topics/shell.rst | 77 +++-- docs/topics/stats.rst | 12 +- 10 files changed, 589 insertions(+), 597 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 6b15a5fbd..33b1a969b 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -252,30 +252,30 @@ The result of running ``response.css('title')`` is a list-like object called and allow you to run further queries to fine-grain the selection or extract the data. -To extract the text from the title above, you can do:: +To extract the text from the title above, you can do: - >>> response.css('title::text').getall() - ['Quotes to Scrape'] +>>> response.css('title::text').getall() +['Quotes to Scrape'] There are two things to note here: one is that we've added ``::text`` to the CSS query, to mean we want to select only the text elements directly inside ```` element. If we don't specify ``::text``, we'd get the full title -element, including its tags:: +element, including its tags: - >>> response.css('title').getall() - ['<title>Quotes to Scrape'] +>>> response.css('title').getall() +['Quotes to Scrape'] The other thing is that the result of calling ``.getall()`` is a list: it is possible that a selector returns more than one result, so we extract them all. -When you know you just want the first result, as in this case, you can do:: +When you know you just want the first result, as in this case, you can do: - >>> response.css('title::text').get() - 'Quotes to Scrape' +>>> response.css('title::text').get() +'Quotes to Scrape' -As an alternative, you could've written:: +As an alternative, you could've written: - >>> response.css('title::text')[0].get() - 'Quotes to Scrape' +>>> response.css('title::text')[0].get() +'Quotes to Scrape' However, using ``.get()`` directly on a :class:`~scrapy.selector.SelectorList` instance avoids an ``IndexError`` and returns ``None`` when it doesn't @@ -288,14 +288,14 @@ to be scraped, you can at least get **some** data. Besides the :meth:`~scrapy.selector.SelectorList.getall` and :meth:`~scrapy.selector.SelectorList.get` methods, you can also use the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular -expressions`_:: +expressions`_: - >>> response.css('title::text').re(r'Quotes.*') - ['Quotes to Scrape'] - >>> response.css('title::text').re(r'Q\w+') - ['Quotes'] - >>> response.css('title::text').re(r'(\w+) to (\w+)') - ['Quotes', 'Scrape'] +>>> response.css('title::text').re(r'Quotes.*') +['Quotes to Scrape'] +>>> response.css('title::text').re(r'Q\w+') +['Quotes'] +>>> response.css('title::text').re(r'(\w+) to (\w+)') +['Quotes', 'Scrape'] In order to find the proper CSS selectors to use, you might find useful opening the response page from the shell in your web browser using ``view(response)``. @@ -312,12 +312,12 @@ visually selected elements, which works in many browsers. XPath: a brief intro ^^^^^^^^^^^^^^^^^^^^ -Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions:: +Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions: - >>> response.xpath('//title') - [] - >>> response.xpath('//title/text()').get() - 'Quotes to Scrape' +>>> response.xpath('//title') +[] +>>> response.xpath('//title/text()').get() +'Quotes to Scrape' XPath expressions are very powerful, and are the foundation of Scrapy Selectors. In fact, CSS selectors are converted to XPath under-the-hood. You @@ -372,35 +372,35 @@ we want:: $ scrapy shell 'http://quotes.toscrape.com' -We get a list of selectors for the quote HTML elements with:: +We get a list of selectors for the quote HTML elements with: - >>> response.css("div.quote") - [, - , - ...] +>>> response.css("div.quote") +[, + , + ...] Each of the selectors returned by the query above allows us to run further queries over their sub-elements. Let's assign the first selector to a -variable, so that we can run our CSS selectors directly on a particular quote:: +variable, so that we can run our CSS selectors directly on a particular quote: - >>> quote = response.css("div.quote")[0] +>>> quote = response.css("div.quote")[0] Now, let's extract ``text``, ``author`` and the ``tags`` from that quote -using the ``quote`` object we just created:: +using the ``quote`` object we just created: - >>> text = quote.css("span.text::text").get() - >>> text - '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”' - >>> author = quote.css("small.author::text").get() - >>> author - 'Albert Einstein' +>>> text = quote.css("span.text::text").get() +>>> text +'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”' +>>> author = quote.css("small.author::text").get() +>>> author +'Albert Einstein' Given that the tags are a list of strings, we can use the ``.getall()`` method -to get all of them:: +to get all of them: - >>> tags = quote.css("div.tags a.tag::text").getall() - >>> tags - ['change', 'deep-thoughts', 'thinking', 'world'] +>>> tags = quote.css("div.tags a.tag::text").getall() +>>> tags +['change', 'deep-thoughts', 'thinking', 'world'] .. invisible-code-block: python @@ -409,16 +409,16 @@ to get all of them:: .. skip: next if(version_info < (3, 6), reason="Only Python 3.6+ dictionaries match the output") Having figured out how to extract each bit, we can now iterate over all the -quotes elements and put them together into a Python dictionary:: +quotes elements and put them together into a Python dictionary: - >>> for quote in response.css("div.quote"): - ... text = quote.css("span.text::text").get() - ... author = quote.css("small.author::text").get() - ... tags = quote.css("div.tags a.tag::text").getall() - ... print(dict(text=text, author=author, tags=tags)) - {'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']} - {'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']} - ... +>>> for quote in response.css("div.quote"): +... text = quote.css("span.text::text").get() +... author = quote.css("small.author::text").get() +... tags = quote.css("div.tags a.tag::text").getall() +... print(dict(text=text, author=author, tags=tags)) +{'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']} +{'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']} +... Extracting data in our spider ----------------------------- @@ -516,23 +516,23 @@ markup: -We can try extracting it in the shell:: +We can try extracting it in the shell: - >>> response.css('li.next a').get() - 'Next ' +>>> response.css('li.next a').get() +'Next ' This gets the anchor element, but we want the attribute ``href``. For that, Scrapy supports a CSS extension that lets you select the attribute contents, -like this:: +like this: - >>> response.css('li.next a::attr(href)').get() - '/page/2/' +>>> response.css('li.next a::attr(href)').get() +'/page/2/' There is also an ``attrib`` property available -(see :ref:`selecting-attributes` for more):: +(see :ref:`selecting-attributes` for more): - >>> response.css('li.next a').attrib['href'] - '/page/2/' +>>> response.css('li.next a').attrib['href'] +'/page/2/' Let's see now our spider modified to recursively follow the link to the next page, extracting data from it:: diff --git a/docs/news.rst b/docs/news.rst index 9dfd28508..217382c57 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -47,13 +47,13 @@ Backward-incompatible changes (:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`) * :class:`~scrapy.loader.ItemLoader` now turns the values of its input item - into lists:: + into lists: - >>> item = MyItem() - >>> item['field'] = 'value1' - >>> loader = ItemLoader(item=item) - >>> item['field'] - ['value1'] + >>> item = MyItem() + >>> item['field'] = 'value1' + >>> loader = ItemLoader(item=item) + >>> item['field'] + ['value1'] This is needed to allow adding values to existing fields (``loader.add_value('field', 'value2')``). diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index bf14643be..1c9315cd8 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -112,13 +112,13 @@ see each quote: With this knowledge we can refine our XPath: Instead of a path to follow, we'll simply select all ``span`` tags with the ``class="text"`` by using -the `has-class-extension`_:: +the `has-class-extension`_: - >>> response.xpath('//span[has-class("text")]/text()').getall() - ['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”, - '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', - '“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', - (...)] +>>> response.xpath('//span[has-class("text")]/text()').getall() +['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', + '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', + '“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”', + ...] And with one simple, cleverer XPath we are able to extract all quotes from the page. We could have constructed a loop over our first XPath to increase diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index 8334ddcec..1c3607860 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -172,27 +172,27 @@ data from it: data in JSON format, which you can then parse with `json.loads`_. For example, if the JavaScript code contains a separate line like - ``var data = {"field": "value"};`` you can extract that data as follows:: + ``var data = {"field": "value"};`` you can extract that data as follows: - >>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n' - >>> json_data = response.css('script::text').re_first(pattern) - >>> json.loads(json_data) - {'field': 'value'} + >>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n' + >>> json_data = response.css('script::text').re_first(pattern) + >>> json.loads(json_data) + {'field': 'value'} - Otherwise, use js2xml_ to convert the JavaScript code into an XML document that you can parse using :ref:`selectors `. For example, if the JavaScript code contains - ``var data = {field: "value"};`` you can extract that data as follows:: + ``var data = {field: "value"};`` you can extract that data as follows: - >>> import js2xml - >>> import lxml.etree - >>> from parsel import Selector - >>> javascript = response.css('script::text').get() - >>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode') - >>> selector = Selector(text=xml) - >>> selector.css('var[name="data"]').get() - 'value' + >>> import js2xml + >>> import lxml.etree + >>> from parsel import Selector + >>> javascript = response.css('script::text').get() + >>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode') + >>> selector = Selector(text=xml) + >>> selector.css('var[name="data"]').get() + 'value' .. _topics-javascript-rendering: diff --git a/docs/topics/items.rst b/docs/topics/items.rst index cdf60208e..15313775b 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -84,77 +84,74 @@ notice the API is very similar to the `dict API`_. Creating items -------------- -:: +>>> product = Product(name='Desktop PC', price=1000) +>>> print(product) +Product(name='Desktop PC', price=1000) - >>> product = Product(name='Desktop PC', price=1000) - >>> print(product) - Product(name='Desktop PC', price=1000) Getting field values -------------------- -:: +>>> product['name'] +Desktop PC +>>> product.get('name') +Desktop PC - >>> product['name'] - Desktop PC - >>> product.get('name') - Desktop PC +>>> product['price'] +1000 - >>> product['price'] - 1000 +>>> product['last_updated'] +Traceback (most recent call last): + ... +KeyError: 'last_updated' - >>> product['last_updated'] - Traceback (most recent call last): - ... - KeyError: 'last_updated' +>>> product.get('last_updated', 'not set') +not set - >>> product.get('last_updated', 'not set') - not set +>>> product['lala'] # getting unknown field +Traceback (most recent call last): + ... +KeyError: 'lala' - >>> product['lala'] # getting unknown field - Traceback (most recent call last): - ... - KeyError: 'lala' +>>> product.get('lala', 'unknown field') +'unknown field' - >>> product.get('lala', 'unknown field') - 'unknown field' +>>> 'name' in product # is name field populated? +True - >>> 'name' in product # is name field populated? - True +>>> 'last_updated' in product # is last_updated populated? +False - >>> 'last_updated' in product # is last_updated populated? - False +>>> 'last_updated' in product.fields # is last_updated a declared field? +True - >>> 'last_updated' in product.fields # is last_updated a declared field? - True +>>> 'lala' in product.fields # is lala a declared field? +False - >>> 'lala' in product.fields # is lala a declared field? - False Setting field values -------------------- -:: +>>> product['last_updated'] = 'today' +>>> product['last_updated'] +today - >>> product['last_updated'] = 'today' - >>> product['last_updated'] - today +>>> product['lala'] = 'test' # setting unknown field +Traceback (most recent call last): + ... +KeyError: 'Product does not support field: lala' - >>> product['lala'] = 'test' # setting unknown field - Traceback (most recent call last): - ... - KeyError: 'Product does not support field: lala' Accessing all populated values ------------------------------ -To access all populated values, just use the typical `dict API`_:: +To access all populated values, just use the typical `dict API`_: - >>> product.keys() - ['price', 'name'] +>>> product.keys() +['price', 'name'] - >>> product.items() - [('price', 1000), ('name', 'Desktop PC')] +>>> product.items() +[('price', 1000), ('name', 'Desktop PC')] .. _copying-items: @@ -194,20 +191,21 @@ To create a deep copy, call :meth:`~scrapy.item.Item.deepcopy` instead Other common tasks ------------------ -Creating dicts from items:: +Creating dicts from items: - >>> dict(product) # create a dict from all populated values - {'price': 1000, 'name': 'Desktop PC'} +>>> dict(product) # create a dict from all populated values +{'price': 1000, 'name': 'Desktop PC'} -Creating items from dicts:: +Creating items from dicts: - >>> Product({'name': 'Laptop PC', 'price': 1500}) - Product(price=1500, name='Laptop PC') +>>> Product({'name': 'Laptop PC', 'price': 1500}) +Product(price=1500, name='Laptop PC') + +>>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict +Traceback (most recent call last): + ... +KeyError: 'Product does not support field: lala' - >>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict - Traceback (most recent call last): - ... - KeyError: 'Product does not support field: lala' Extending Items =============== diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 793636f59..87d9d262f 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -132,21 +132,21 @@ and check the code of the spider to discover the nasty line that is generating the leaks (passing response references inside requests). Sometimes extra information about live objects can be helpful. -Let's check the oldest response:: +Let's check the oldest response: - >>> from scrapy.utils.trackref import get_oldest - >>> r = get_oldest('HtmlResponse') - >>> r.url - 'http://www.somenastyspider.com/product.php?pid=123' +>>> from scrapy.utils.trackref import get_oldest +>>> r = get_oldest('HtmlResponse') +>>> r.url +'http://www.somenastyspider.com/product.php?pid=123' If you want to iterate over all objects, instead of getting the oldest one, you -can use the :func:`scrapy.utils.trackref.iter_all` function:: +can use the :func:`scrapy.utils.trackref.iter_all` function: - >>> from scrapy.utils.trackref import iter_all - >>> [r.url for r in iter_all('HtmlResponse')] - ['http://www.somenastyspider.com/product.php?pid=123', - 'http://www.somenastyspider.com/product.php?pid=584', - ... +>>> from scrapy.utils.trackref import iter_all +>>> [r.url for r in iter_all('HtmlResponse')] +['http://www.somenastyspider.com/product.php?pid=123', + 'http://www.somenastyspider.com/product.php?pid=584', +...] Too many spiders? ----------------- @@ -155,10 +155,10 @@ If your project has too many spiders executed in parallel, the output of :func:`prefs()` can be difficult to read. For this reason, that function has a ``ignore`` argument which can be used to ignore a particular class (and all its subclases). For -example, this won't show any live references to spiders:: +example, this won't show any live references to spiders: - >>> from scrapy.spiders import Spider - >>> prefs(ignore=Spider) +>>> from scrapy.spiders import Spider +>>> prefs(ignore=Spider) .. module:: scrapy.utils.trackref :synopsis: Track references of live objects @@ -214,41 +214,41 @@ If you use ``pip``, you can install Guppy with the following command:: The telnet console also comes with a built-in shortcut (``hpy``) for accessing Guppy heap objects. Here's an example to view all Python objects available in -the heap using Guppy:: +the heap using Guppy: - >>> x = hpy.heap() - >>> x.bytype - Partition of a set of 297033 objects. Total size = 52587824 bytes. - Index Count % Size % Cumulative % Type - 0 22307 8 16423880 31 16423880 31 dict - 1 122285 41 12441544 24 28865424 55 str - 2 68346 23 5966696 11 34832120 66 tuple - 3 227 0 5836528 11 40668648 77 unicode - 4 2461 1 2222272 4 42890920 82 type - 5 16870 6 2024400 4 44915320 85 function - 6 13949 5 1673880 3 46589200 89 types.CodeType - 7 13422 5 1653104 3 48242304 92 list - 8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern - 9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers - <1676 more rows. Type e.g. '_.more' to view.> +>>> x = hpy.heap() +>>> x.bytype +Partition of a set of 297033 objects. Total size = 52587824 bytes. + Index Count % Size % Cumulative % Type + 0 22307 8 16423880 31 16423880 31 dict + 1 122285 41 12441544 24 28865424 55 str + 2 68346 23 5966696 11 34832120 66 tuple + 3 227 0 5836528 11 40668648 77 unicode + 4 2461 1 2222272 4 42890920 82 type + 5 16870 6 2024400 4 44915320 85 function + 6 13949 5 1673880 3 46589200 89 types.CodeType + 7 13422 5 1653104 3 48242304 92 list + 8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern + 9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers +<1676 more rows. Type e.g. '_.more' to view.> You can see that most space is used by dicts. Then, if you want to see from -which attribute those dicts are referenced, you could do:: +which attribute those dicts are referenced, you could do: - >>> x.bytype[0].byvia - Partition of a set of 22307 objects. Total size = 16423880 bytes. - Index Count % Size % Cumulative % Referred Via: - 0 10982 49 9416336 57 9416336 57 '.__dict__' - 1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals' - 2 3097 14 1122904 7 13220744 80 - 3 990 4 277200 2 13497944 82 "['cookies']" - 4 987 4 276360 2 13774304 84 "['cache']" - 5 985 4 275800 2 14050104 86 "['meta']" - 6 897 4 251160 2 14301264 87 '[2]' - 7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']" - 8 672 3 188160 1 14686312 89 "['cb_kwargs']" - 9 27 0 155016 1 14841328 90 '[1]' - <333 more rows. Type e.g. '_.more' to view.> +>>> x.bytype[0].byvia +Partition of a set of 22307 objects. Total size = 16423880 bytes. + Index Count % Size % Cumulative % Referred Via: + 0 10982 49 9416336 57 9416336 57 '.__dict__' + 1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals' + 2 3097 14 1122904 7 13220744 80 + 3 990 4 277200 2 13497944 82 "['cookies']" + 4 987 4 276360 2 13774304 84 "['cache']" + 5 985 4 275800 2 14050104 86 "['meta']" + 6 897 4 251160 2 14301264 87 '[2]' + 7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']" + 8 672 3 188160 1 14686312 89 "['cb_kwargs']" + 9 27 0 155016 1 14841328 90 '[1]' +<333 more rows. Type e.g. '_.more' to view.> As you can see, the Guppy module is very powerful but also requires some deep knowledge about Python internals. For more info about Guppy, refer to the @@ -269,32 +269,32 @@ If you use ``pip``, you can install muppy with the following command:: pip install Pympler Here's an example to view all Python objects available in -the heap using muppy:: +the heap using muppy: - >>> from pympler import muppy - >>> all_objects = muppy.get_objects() - >>> len(all_objects) - 28667 - >>> from pympler import summary - >>> suml = summary.summarize(all_objects) - >>> summary.print_(suml) - types | # objects | total size - ==================================== | =========== | ============ - >> from pympler import muppy +>>> all_objects = muppy.get_objects() +>>> len(all_objects) +28667 +>>> from pympler import summary +>>> suml = summary.summarize(all_objects) +>>> summary.print_(suml) + types | # objects | total size +==================================== | =========== | ============ + >> from scrapy.loader import ItemLoader - >>> il = ItemLoader(item=Product()) - >>> il.add_value('name', [u'Welcome to my', u'website']) - >>> il.add_value('price', [u'€', u'1000']) - >>> il.load_item() - {'name': u'Welcome to my website', 'price': u'1000'} +>>> from scrapy.loader import ItemLoader +>>> il = ItemLoader(item=Product()) +>>> il.add_value('name', [u'Welcome to my', u'website']) +>>> il.add_value('price', [u'€', u'1000']) +>>> il.load_item() +{'name': u'Welcome to my website', 'price': u'1000'} The precedence order, for both input and output processors, is as follows: @@ -314,11 +312,11 @@ ItemLoader objects applied before processors :type re: str or compiled regex - Examples:: + Examples: - >>> from scrapy.loader.processors import TakeFirst - >>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)') - 'FOO` + >>> from scrapy.loader.processors import TakeFirst + >>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)') + 'FOO` .. method:: add_value(field_name, value, \*processors, \**kwargs) @@ -639,12 +637,12 @@ Here is a list of all built-in processors: values unchanged. It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts. - Example:: + Example: - >>> from scrapy.loader.processors import Identity - >>> proc = Identity() - >>> proc(['one', 'two', 'three']) - ['one', 'two', 'three'] + >>> from scrapy.loader.processors import Identity + >>> proc = Identity() + >>> proc(['one', 'two', 'three']) + ['one', 'two', 'three'] .. class:: TakeFirst @@ -652,12 +650,12 @@ Here is a list of all built-in processors: so it's typically used as an output processor to single-valued fields. It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts. - Example:: + Example: - >>> from scrapy.loader.processors import TakeFirst - >>> proc = TakeFirst() - >>> proc(['', 'one', 'two', 'three']) - 'one' + >>> from scrapy.loader.processors import TakeFirst + >>> proc = TakeFirst() + >>> proc(['', 'one', 'two', 'three']) + 'one' .. class:: Join(separator=u' ') @@ -667,15 +665,15 @@ Here is a list of all built-in processors: When using the default separator, this processor is equivalent to the function: ``u' '.join`` - Examples:: + Examples: - >>> from scrapy.loader.processors import Join - >>> proc = Join() - >>> proc(['one', 'two', 'three']) - 'one two three' - >>> proc = Join('
') - >>> proc(['one', 'two', 'three']) - 'one
two
three' + >>> from scrapy.loader.processors import Join + >>> proc = Join() + >>> proc(['one', 'two', 'three']) + 'one two three' + >>> proc = Join('
') + >>> proc(['one', 'two', 'three']) + 'one
two
three' .. class:: Compose(\*functions, \**default_loader_context) @@ -688,12 +686,12 @@ Here is a list of all built-in processors: By default, stop process on ``None`` value. This behaviour can be changed by passing keyword argument ``stop_on_none=False``. - Example:: + Example: - >>> from scrapy.loader.processors import Compose - >>> proc = Compose(lambda v: v[0], str.upper) - >>> proc(['hello', 'world']) - 'HELLO' + >>> from scrapy.loader.processors import Compose + >>> proc = Compose(lambda v: v[0], str.upper) + >>> proc(['hello', 'world']) + 'HELLO' Each function can optionally receive a ``loader_context`` parameter. For those which do, this processor will pass the currently active :ref:`Loader @@ -732,15 +730,15 @@ Here is a list of all built-in processors: :meth:`~scrapy.selector.Selector.extract` method of :ref:`selectors `, which returns a list of unicode strings. - The example below should clarify how it works:: + The example below should clarify how it works: - >>> def filter_world(x): - ... return None if x == 'world' else x - ... - >>> from scrapy.loader.processors import MapCompose - >>> proc = MapCompose(filter_world, str.upper) - >>> proc(['hello', 'world', 'this', 'is', 'scrapy']) - ['HELLO, 'THIS', 'IS', 'SCRAPY'] + >>> def filter_world(x): + ... return None if x == 'world' else x + ... + >>> from scrapy.loader.processors import MapCompose + >>> proc = MapCompose(filter_world, str.upper) + >>> proc(['hello', 'world', 'this', 'is', 'scrapy']) + ['HELLO, 'THIS', 'IS', 'SCRAPY'] As with the Compose processor, functions can receive Loader contexts, and ``__init__`` method keyword arguments are used as default context values. See @@ -752,21 +750,21 @@ Here is a list of all built-in processors: Requires jmespath (https://github.com/jmespath/jmespath.py) to run. This processor takes only one input at a time. - Example:: + Example: - >>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose - >>> proc = SelectJmes("foo") #for direct use on lists and dictionaries - >>> proc({'foo': 'bar'}) - 'bar' - >>> proc({'foo': {'bar': 'baz'}}) - {'bar': 'baz'} + >>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose + >>> proc = SelectJmes("foo") #for direct use on lists and dictionaries + >>> proc({'foo': 'bar'}) + 'bar' + >>> proc({'foo': {'bar': 'baz'}}) + {'bar': 'baz'} - Working with Json:: + Working with Json: - >>> import json - >>> proc_single_json_str = Compose(json.loads, SelectJmes("foo")) - >>> proc_single_json_str('{"foo": "bar"}') - 'bar' - >>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo'))) - >>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]') - ['bar'] + >>> import json + >>> proc_single_json_str = Compose(json.loads, SelectJmes("foo")) + >>> proc_single_json_str('{"foo": "bar"}') + 'bar' + >>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo'))) + >>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]') + ['bar'] diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 282a585d4..8ec758b0e 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -51,18 +51,18 @@ Constructing selectors .. highlight:: python Response objects expose a :class:`~scrapy.selector.Selector` instance -on ``.selector`` attribute:: +on ``.selector`` attribute: - >>> response.selector.xpath('//span/text()').get() - 'good' +>>> response.selector.xpath('//span/text()').get() +'good' Querying responses using XPath and CSS is so common that responses include two -more shortcuts: ``response.xpath()`` and ``response.css()``:: +more shortcuts: ``response.xpath()`` and ``response.css()``: - >>> response.xpath('//span/text()').get() - 'good' - >>> response.css('span::text').get() - 'good' +>>> response.xpath('//span/text()').get() +'good' +>>> response.css('span::text').get() +'good' Scrapy selectors are instances of :class:`~scrapy.selector.Selector` class constructed by passing either :class:`~scrapy.http.TextResponse` object or @@ -74,21 +74,21 @@ shortcuts. By using ``response.selector`` or one of these shortcuts you can also ensure the response body is parsed only once. But if required, it is possible to use ``Selector`` directly. -Constructing from text:: +Constructing from text: - >>> from scrapy.selector import Selector - >>> body = 'good' - >>> Selector(text=body).xpath('//span/text()').get() - 'good' +>>> from scrapy.selector import Selector +>>> body = 'good' +>>> Selector(text=body).xpath('//span/text()').get() +'good' Constructing from response - :class:`~scrapy.http.HtmlResponse` is one of -:class:`~scrapy.http.TextResponse` subclasses:: +:class:`~scrapy.http.TextResponse` subclasses: - >>> from scrapy.selector import Selector - >>> from scrapy.http import HtmlResponse - >>> response = HtmlResponse(url='http://example.com', body=body) - >>> Selector(response=response).xpath('//span/text()').get() - 'good' +>>> from scrapy.selector import Selector +>>> from scrapy.http import HtmlResponse +>>> response = HtmlResponse(url='http://example.com', body=body) +>>> Selector(response=response).xpath('//span/text()').get() +'good' ``Selector`` automatically chooses the best parsing rules (XML vs HTML) based on input type. @@ -123,118 +123,118 @@ Since we're dealing with HTML, the selector will automatically use an HTML parse .. highlight:: python So, by looking at the :ref:`HTML code ` of that -page, let's construct an XPath for selecting the text inside the title tag:: +page, let's construct an XPath for selecting the text inside the title tag: - >>> response.xpath('//title/text()') - [] +>>> response.xpath('//title/text()') +[] To actually extract the textual data, you must call the selector ``.get()`` -or ``.getall()`` methods, as follows:: +or ``.getall()`` methods, as follows: - >>> response.xpath('//title/text()').getall() - ['Example website'] - >>> response.xpath('//title/text()').get() - 'Example website' +>>> response.xpath('//title/text()').getall() +['Example website'] +>>> response.xpath('//title/text()').get() +'Example website' ``.get()`` always returns a single result; if there are several matches, content of a first match is returned; if there are no matches, None is returned. ``.getall()`` returns a list with all results. Notice that CSS selectors can select text or attribute nodes using CSS3 -pseudo-elements:: +pseudo-elements: - >>> response.css('title::text').get() - 'Example website' +>>> response.css('title::text').get() +'Example website' As you can see, ``.xpath()`` and ``.css()`` methods return a :class:`~scrapy.selector.SelectorList` instance, which is a list of new -selectors. This API can be used for quickly selecting nested data:: +selectors. This API can be used for quickly selecting nested data: - >>> response.css('img').xpath('@src').getall() - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> response.css('img').xpath('@src').getall() +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] If you want to extract only the first matched element, you can call the selector ``.get()`` (or its alias ``.extract_first()`` commonly used in -previous Scrapy versions):: +previous Scrapy versions): - >>> response.xpath('//div[@id="images"]/a/text()').get() - 'Name: My image 1 ' +>>> response.xpath('//div[@id="images"]/a/text()').get() +'Name: My image 1 ' -It returns ``None`` if no element was found:: +It returns ``None`` if no element was found: - >>> response.xpath('//div[@id="not-exists"]/text()').get() is None - True +>>> response.xpath('//div[@id="not-exists"]/text()').get() is None +True A default return value can be provided as an argument, to be used instead of ``None``: - >>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found') - 'not-found' +>>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found') +'not-found' Instead of using e.g. ``'@src'`` XPath it is possible to query for attributes -using ``.attrib`` property of a :class:`~scrapy.selector.Selector`:: +using ``.attrib`` property of a :class:`~scrapy.selector.Selector`: - >>> [img.attrib['src'] for img in response.css('img')] - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> [img.attrib['src'] for img in response.css('img')] +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] As a shortcut, ``.attrib`` is also available on SelectorList directly; -it returns attributes for the first matching element:: +it returns attributes for the first matching element: - >>> response.css('img').attrib['src'] - 'image1_thumb.jpg' +>>> response.css('img').attrib['src'] +'image1_thumb.jpg' This is most useful when only a single result is expected, e.g. when selecting -by id, or selecting unique elements on a web page:: +by id, or selecting unique elements on a web page: - >>> response.css('base').attrib['href'] - 'http://example.com/' +>>> response.css('base').attrib['href'] +'http://example.com/' -Now we're going to get the base URL and some image links:: +Now we're going to get the base URL and some image links: - >>> response.xpath('//base/@href').get() - 'http://example.com/' +>>> response.xpath('//base/@href').get() +'http://example.com/' - >>> response.css('base::attr(href)').get() - 'http://example.com/' +>>> response.css('base::attr(href)').get() +'http://example.com/' - >>> response.css('base').attrib['href'] - 'http://example.com/' +>>> response.css('base').attrib['href'] +'http://example.com/' - >>> response.xpath('//a[contains(@href, "image")]/@href').getall() - ['image1.html', - 'image2.html', - 'image3.html', - 'image4.html', - 'image5.html'] +>>> response.xpath('//a[contains(@href, "image")]/@href').getall() +['image1.html', + 'image2.html', + 'image3.html', + 'image4.html', + 'image5.html'] - >>> response.css('a[href*=image]::attr(href)').getall() - ['image1.html', - 'image2.html', - 'image3.html', - 'image4.html', - 'image5.html'] +>>> response.css('a[href*=image]::attr(href)').getall() +['image1.html', + 'image2.html', + 'image3.html', + 'image4.html', + 'image5.html'] - >>> response.xpath('//a[contains(@href, "image")]/img/@src').getall() - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> response.xpath('//a[contains(@href, "image")]/img/@src').getall() +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] - >>> response.css('a[href*=image] img::attr(src)').getall() - ['image1_thumb.jpg', - 'image2_thumb.jpg', - 'image3_thumb.jpg', - 'image4_thumb.jpg', - 'image5_thumb.jpg'] +>>> response.css('a[href*=image] img::attr(src)').getall() +['image1_thumb.jpg', + 'image2_thumb.jpg', + 'image3_thumb.jpg', + 'image4_thumb.jpg', + 'image5_thumb.jpg'] .. _topics-selectors-css-extensions: @@ -259,47 +259,47 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**: Examples: -* ``title::text`` selects children text nodes of a descendant ```` element:: +* ``title::text`` selects children text nodes of a descendant ``<title>`` element: - >>> response.css('title::text').get() - 'Example website' +>>> response.css('title::text').get() +'Example website' -* ``*::text`` selects all descendant text nodes of the current selector context:: +* ``*::text`` selects all descendant text nodes of the current selector context: - >>> response.css('#images *::text').getall() - ['\n ', - 'Name: My image 1 ', - '\n ', - 'Name: My image 2 ', - '\n ', - 'Name: My image 3 ', - '\n ', - 'Name: My image 4 ', - '\n ', - 'Name: My image 5 ', - '\n '] +>>> response.css('#images *::text').getall() +['\n ', + 'Name: My image 1 ', + '\n ', + 'Name: My image 2 ', + '\n ', + 'Name: My image 3 ', + '\n ', + 'Name: My image 4 ', + '\n ', + 'Name: My image 5 ', + '\n '] * ``foo::text`` returns no results if ``foo`` element exists, but contains - no text (i.e. text is empty):: + no text (i.e. text is empty): - >>> response.css('img::text').getall() - [] +>>> response.css('img::text').getall() +[] This means ``.css('foo::text').get()`` could return None even if an element - exists. Use ``default=''`` if you always want a string:: + exists. Use ``default=''`` if you always want a string: - >>> response.css('img::text').get() - >>> response.css('img::text').get(default='') - '' +>>> response.css('img::text').get() +>>> response.css('img::text').get(default='') +'' -* ``a::attr(href)`` selects the *href* attribute value of descendant links:: +* ``a::attr(href)`` selects the *href* attribute value of descendant links: - >>> response.css('a::attr(href)').getall() - ['image1.html', - 'image2.html', - 'image3.html', - 'image4.html', - 'image5.html'] +>>> response.css('a::attr(href)').getall() +['image1.html', + 'image2.html', + 'image3.html', + 'image4.html', + 'image5.html'] .. note:: See also: :ref:`selecting-attributes`. @@ -318,25 +318,24 @@ Nesting selectors The selection methods (``.xpath()`` or ``.css()``) return a list of selectors of the same type, so you can call the selection methods for those selectors -too. Here's an example:: +too. Here's an example: - >>> links = response.xpath('//a[contains(@href, "image")]') - >>> links.getall() - ['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>', - '<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>', - '<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>', - '<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>', - '<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>'] +>>> links = response.xpath('//a[contains(@href, "image")]') +>>> links.getall() +['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>', + '<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>', + '<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>', + '<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>', + '<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>'] - >>> for index, link in enumerate(links): - ... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get()) - ... print('Link number %d points to url %r and image %r' % args) - - Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg' - Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg' - Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg' - Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg' - Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg' +>>> for index, link in enumerate(links): +... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get()) +... print('Link number %d points to url %r and image %r' % args) +Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg' +Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg' +Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg' +Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg' +Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg' .. _selecting-attributes: @@ -344,42 +343,42 @@ Selecting element attributes ---------------------------- There are several ways to get a value of an attribute. First, one can use -XPath syntax:: +XPath syntax: - >>> response.xpath("//a/@href").getall() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] +>>> response.xpath("//a/@href").getall() +['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] XPath syntax has a few advantages: it is a standard XPath feature, and ``@attributes`` can be used in other parts of an XPath expression - e.g. it is possible to filter by attribute value. Scrapy also provides an extension to CSS selectors (``::attr(...)``) -which allows to get attribute values:: +which allows to get attribute values: - >>> response.css('a::attr(href)').getall() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] +>>> response.css('a::attr(href)').getall() +['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] In addition to that, there is a ``.attrib`` property of Selector. You can use it if you prefer to lookup attributes in Python -code, without using XPaths or CSS extensions:: +code, without using XPaths or CSS extensions: - >>> [a.attrib['href'] for a in response.css('a')] - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] +>>> [a.attrib['href'] for a in response.css('a')] +['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] This property is also available on SelectorList; it returns a dictionary with attributes of a first matching element. It is convenient to use when a selector is expected to give a single result (e.g. when selecting by element -ID, or when selecting an unique element on a page):: +ID, or when selecting an unique element on a page): - >>> response.css('base').attrib - {'href': 'http://example.com/'} - >>> response.css('base').attrib['href'] - 'http://example.com/' +>>> response.css('base').attrib +{'href': 'http://example.com/'} +>>> response.css('base').attrib['href'] +'http://example.com/' -``.attrib`` property of an empty SelectorList is empty:: +``.attrib`` property of an empty SelectorList is empty: - >>> response.css('foo').attrib - {} +>>> response.css('foo').attrib +{} Using selectors with regular expressions ---------------------------------------- @@ -390,21 +389,21 @@ data using regular expressions. However, unlike using ``.xpath()`` or can't construct nested ``.re()`` calls. Here's an example used to extract image names from the :ref:`HTML code -<topics-selectors-htmlcode>` above:: +<topics-selectors-htmlcode>` above: - >>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)') - ['My image 1', - 'My image 2', - 'My image 3', - 'My image 4', - 'My image 5'] +>>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)') +['My image 1', + 'My image 2', + 'My image 3', + 'My image 4', + 'My image 5'] There's an additional helper reciprocating ``.get()`` (and its alias ``.extract_first()``) for ``.re()``, named ``.re_first()``. -Use it to extract just the first matching string:: +Use it to extract just the first matching string: - >>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)') - 'My image 1' +>>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)') +'My image 1' .. _old-extraction-api: @@ -422,28 +421,28 @@ and readable code. The following examples show how these methods map to each other. -1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``:: +1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``: - >>> response.css('a::attr(href)').get() - 'image1.html' - >>> response.css('a::attr(href)').extract_first() - 'image1.html' + >>> response.css('a::attr(href)').get() + 'image1.html' + >>> response.css('a::attr(href)').extract_first() + 'image1.html' -2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``:: +2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``: - >>> response.css('a::attr(href)').getall() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] - >>> response.css('a::attr(href)').extract() - ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] + >>> response.css('a::attr(href)').getall() + ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] + >>> response.css('a::attr(href)').extract() + ['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html'] -3. ``Selector.get()`` is the same as ``Selector.extract()``:: +3. ``Selector.get()`` is the same as ``Selector.extract()``: - >>> response.css('a::attr(href)')[0].get() - 'image1.html' - >>> response.css('a::attr(href)')[0].extract() - 'image1.html' + >>> response.css('a::attr(href)')[0].get() + 'image1.html' + >>> response.css('a::attr(href)')[0].extract() + 'image1.html' -4. For consistency, there is also ``Selector.getall()``, which returns a list:: +4. For consistency, there is also ``Selector.getall()``, which returns a list: >>> response.css('a::attr(href)')[0].getall() ['image1.html'] @@ -481,26 +480,26 @@ with ``/``, that XPath will be absolute to the document and not relative to the ``Selector`` you're calling it from. For example, suppose you want to extract all ``<p>`` elements inside ``<div>`` -elements. First, you would get all ``<div>`` elements:: +elements. First, you would get all ``<div>`` elements: - >>> divs = response.xpath('//div') +>>> divs = response.xpath('//div') At first, you may be tempted to use the following approach, which is wrong, as it actually extracts all ``<p>`` elements from the document, not only those -inside ``<div>`` elements:: +inside ``<div>`` elements: - >>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document - ... print(p.get()) +>>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document +... print(p.get()) -This is the proper way to do it (note the dot prefixing the ``.//p`` XPath):: +This is the proper way to do it (note the dot prefixing the ``.//p`` XPath): - >>> for p in divs.xpath('.//p'): # extracts all <p> inside - ... print(p.get()) +>>> for p in divs.xpath('.//p'): # extracts all <p> inside +... print(p.get()) -Another common case would be to extract all direct ``<p>`` children:: +Another common case would be to extract all direct ``<p>`` children: - >>> for p in divs.xpath('p'): - ... print(p.get()) +>>> for p in divs.xpath('p'): +... print(p.get()) For more details about relative XPaths see the `Location Paths`_ section in the XPath specification. @@ -521,12 +520,12 @@ for that you may end up with more elements that you want, if they have a differe class name that shares the string ``someclass``. As it turns out, Scrapy selectors allow you to chain selectors, so most of the time -you can just select by class using CSS and then switch to XPath when needed:: +you can just select by class using CSS and then switch to XPath when needed: - >>> from scrapy import Selector - >>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>') - >>> sel.css('.shout').xpath('./time/@datetime').getall() - ['2014-07-23 19:00'] +>>> from scrapy import Selector +>>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>') +>>> sel.css('.shout').xpath('./time/@datetime').getall() +['2014-07-23 19:00'] This is cleaner than using the verbose XPath trick shown above. Just remember to use the ``.`` in the XPath expressions that will follow. @@ -538,41 +537,41 @@ Beware of the difference between //node[1] and (//node)[1] ``(//node)[1]`` selects all the nodes in the document, and then gets only the first of them. -Example:: +Example: - >>> from scrapy import Selector - >>> sel = Selector(text=""" - ....: <ul class="list"> - ....: <li>1</li> - ....: <li>2</li> - ....: <li>3</li> - ....: </ul> - ....: <ul class="list"> - ....: <li>4</li> - ....: <li>5</li> - ....: <li>6</li> - ....: </ul>""") - >>> xp = lambda x: sel.xpath(x).getall() +>>> from scrapy import Selector +>>> sel = Selector(text=""" +....: <ul class="list"> +....: <li>1</li> +....: <li>2</li> +....: <li>3</li> +....: </ul> +....: <ul class="list"> +....: <li>4</li> +....: <li>5</li> +....: <li>6</li> +....: </ul>""") +>>> xp = lambda x: sel.xpath(x).getall() -This gets all first ``<li>`` elements under whatever it is its parent:: +This gets all first ``<li>`` elements under whatever it is its parent: - >>> xp("//li[1]") - ['<li>1</li>', '<li>4</li>'] +>>> xp("//li[1]") +['<li>1</li>', '<li>4</li>'] -And this gets the first ``<li>`` element in the whole document:: +And this gets the first ``<li>`` element in the whole document: - >>> xp("(//li)[1]") - ['<li>1</li>'] +>>> xp("(//li)[1]") +['<li>1</li>'] -This gets all first ``<li>`` elements under an ``<ul>`` parent:: +This gets all first ``<li>`` elements under an ``<ul>`` parent: - >>> xp("//ul/li[1]") - ['<li>1</li>', '<li>4</li>'] +>>> xp("//ul/li[1]") +['<li>1</li>', '<li>4</li>'] -And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document:: +And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document: - >>> xp("(//ul/li)[1]") - ['<li>1</li>'] +>>> xp("(//ul/li)[1]") +['<li>1</li>'] Using text nodes in a condition ------------------------------- @@ -584,34 +583,34 @@ This is because the expression ``.//text()`` yields a collection of text element And when a node-set is converted to a string, which happens when it is passed as argument to a string function like ``contains()`` or ``starts-with()``, it results in the text for the first element only. -Example:: +Example: - >>> from scrapy import Selector - >>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>') +>>> from scrapy import Selector +>>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>') -Converting a *node-set* to string:: +Converting a *node-set* to string: - >>> sel.xpath('//a//text()').getall() # take a peek at the node-set - ['Click here to go to the ', 'Next Page'] - >>> sel.xpath("string(//a[1]//text())").getall() # convert it to string - ['Click here to go to the '] +>>> sel.xpath('//a//text()').getall() # take a peek at the node-set +['Click here to go to the ', 'Next Page'] +>>> sel.xpath("string(//a[1]//text())").getall() # convert it to string +['Click here to go to the '] -A *node* converted to a string, however, puts together the text of itself plus of all its descendants:: +A *node* converted to a string, however, puts together the text of itself plus of all its descendants: - >>> sel.xpath("//a[1]").getall() # select the first node - ['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] - >>> sel.xpath("string(//a[1])").getall() # convert it to string - ['Click here to go to the Next Page'] +>>> sel.xpath("//a[1]").getall() # select the first node +['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] +>>> sel.xpath("string(//a[1])").getall() # convert it to string +['Click here to go to the Next Page'] -So, using the ``.//text()`` node-set won't select anything in this case:: +So, using the ``.//text()`` node-set won't select anything in this case: - >>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall() - [] +>>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall() +[] -But using the ``.`` to mean the node, works:: +But using the ``.`` to mean the node, works: - >>> sel.xpath("//a[contains(., 'Next Page')]").getall() - ['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] +>>> sel.xpath("//a[contains(., 'Next Page')]").getall() +['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] .. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions @@ -627,17 +626,17 @@ some arguments in your queries with placeholders like ``?``, which are then substituted with values passed with the query. Here's an example to match an element based on its "id" attribute value, -without hard-coding it (that was shown previously):: +without hard-coding it (that was shown previously): - >>> # `$val` used in the expression, a `val` argument needs to be passed - >>> response.xpath('//div[@id=$val]/a/text()', val='images').get() - 'Name: My image 1 ' +>>> # `$val` used in the expression, a `val` argument needs to be passed +>>> response.xpath('//div[@id=$val]/a/text()', val='images').get() +'Name: My image 1 ' Here's another example, to find the "id" attribute of a ``<div>`` tag containing -five ``<a>`` children (here we pass the value ``5`` as an integer):: +five ``<a>`` children (here we pass the value ``5`` as an integer): - >>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get() - 'images' +>>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get() +'images' All variable references must have a binding value when calling ``.xpath()`` (otherwise you'll get a ``ValueError: XPath error:`` exception). @@ -687,19 +686,19 @@ You can see several namespace declarations including a default .. highlight:: python Once in the shell we can try selecting all ``<link>`` objects and see that it -doesn't work (because the Atom XML namespace is obfuscating those nodes):: +doesn't work (because the Atom XML namespace is obfuscating those nodes): - >>> response.xpath("//link") - [] +>>> response.xpath("//link") +[] But once we call the :meth:`Selector.remove_namespaces` method, all -nodes can be accessed directly by their names:: +nodes can be accessed directly by their names: - >>> response.selector.remove_namespaces() - >>> response.xpath("//link") - [<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>, - <Selector xpath='//link' data='<link rel="next" type="application/atom+'>, - ... +>>> response.selector.remove_namespaces() +>>> response.xpath("//link") +[<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>, + <Selector xpath='//link' data='<link rel="next" type="application/atom+'>, + ... If you wonder why the namespace removal procedure isn't always called by default instead of having to call it manually, this is because of two reasons, which, in order @@ -734,26 +733,25 @@ Regular expressions The ``test()`` function, for example, can prove quite useful when XPath's ``starts-with()`` or ``contains()`` are not sufficient. -Example selecting links in list item with a "class" attribute ending with a digit:: +Example selecting links in list item with a "class" attribute ending with a digit: - >>> from scrapy import Selector - >>> doc = u""" - ... <div> - ... <ul> - ... <li class="item-0"><a href="link1.html">first item</a></li> - ... <li class="item-1"><a href="link2.html">second item</a></li> - ... <li class="item-inactive"><a href="link3.html">third item</a></li> - ... <li class="item-1"><a href="link4.html">fourth item</a></li> - ... <li class="item-0"><a href="link5.html">fifth item</a></li> - ... </ul> - ... </div> - ... """ - >>> sel = Selector(text=doc, type="html") - >>> sel.xpath('//li//@href').getall() - ['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html'] - >>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall() - ['link1.html', 'link2.html', 'link4.html', 'link5.html'] - >>> +>>> from scrapy import Selector +>>> doc = u""" +... <div> +... <ul> +... <li class="item-0"><a href="link1.html">first item</a></li> +... <li class="item-1"><a href="link2.html">second item</a></li> +... <li class="item-inactive"><a href="link3.html">third item</a></li> +... <li class="item-1"><a href="link4.html">fourth item</a></li> +... <li class="item-0"><a href="link5.html">fifth item</a></li> +... </ul> +... </div> +... """ +>>> sel = Selector(text=doc, type="html") +>>> sel.xpath('//li//@href').getall() +['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html'] +>>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall() +['link1.html', 'link2.html', 'link4.html', 'link5.html'] .. warning:: C library ``libxslt`` doesn't natively support EXSLT regular expressions so `lxml`_'s implementation uses hooks to Python's ``re`` module. @@ -849,7 +847,6 @@ with groups of itemscopes and corresponding itemprops:: current scope: ['http://schema.org/Rating'] properties: ['worstRating', 'ratingValue', 'bestRating'] - >>> Here we first iterate over ``itemscope`` elements, and for each one, we look for all ``itemprops`` elements and exclude those that are themselves @@ -877,15 +874,15 @@ For the following HTML:: .. highlight:: python -You can use it like this:: +You can use it like this: - >>> response.xpath('//p[has-class("foo")]') - [<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>, - <Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>] - >>> response.xpath('//p[has-class("foo", "bar-baz")]') - [<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>] - >>> response.xpath('//p[has-class("foo", "bar")]') - [] +>>> response.xpath('//p[has-class("foo")]') +[<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>, + <Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>] +>>> response.xpath('//p[has-class("foo", "bar-baz")]') +[<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>] +>>> response.xpath('//p[has-class("foo", "bar")]') +[] So XPath ``//p[has-class("foo", "bar-baz")]`` is roughly equivalent to CSS ``p.foo.bar-baz``. Please note, that it is slower in most of the cases, diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index 68a0b19b5..c1fdfd221 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -177,47 +177,46 @@ all start with the ``[s]`` prefix):: >>> -After that, we can start playing with the objects:: +After that, we can start playing with the objects: - >>> response.xpath('//title/text()').get() - 'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' +>>> response.xpath('//title/text()').get() +'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' - >>> fetch("https://reddit.com") +>>> fetch("https://reddit.com") - >>> response.xpath('//title/text()').get() - 'reddit: the front page of the internet' +>>> response.xpath('//title/text()').get() +'reddit: the front page of the internet' - >>> request = request.replace(method="POST") +>>> request = request.replace(method="POST") - >>> fetch(request) +>>> fetch(request) - >>> response.status - 404 +>>> response.status +404 - >>> from pprint import pprint +>>> from pprint import pprint - >>> pprint(response.headers) - {'Accept-Ranges': ['bytes'], - 'Cache-Control': ['max-age=0, must-revalidate'], - 'Content-Type': ['text/html; charset=UTF-8'], - 'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'], - 'Server': ['snooserv'], - 'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', - 'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', - 'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', - 'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'], - 'Vary': ['accept-encoding'], - 'Via': ['1.1 varnish'], - 'X-Cache': ['MISS'], - 'X-Cache-Hits': ['0'], - 'X-Content-Type-Options': ['nosniff'], - 'X-Frame-Options': ['SAMEORIGIN'], - 'X-Moose': ['majestic'], - 'X-Served-By': ['cache-cdg8730-CDG'], - 'X-Timer': ['S1481214079.394283,VS0,VE159'], - 'X-Ua-Compatible': ['IE=edge'], - 'X-Xss-Protection': ['1; mode=block']} - >>> +>>> pprint(response.headers) +{'Accept-Ranges': ['bytes'], + 'Cache-Control': ['max-age=0, must-revalidate'], + 'Content-Type': ['text/html; charset=UTF-8'], + 'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'], + 'Server': ['snooserv'], + 'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', + 'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', + 'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure', + 'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'], + 'Vary': ['accept-encoding'], + 'Via': ['1.1 varnish'], + 'X-Cache': ['MISS'], + 'X-Cache-Hits': ['0'], + 'X-Content-Type-Options': ['nosniff'], + 'X-Frame-Options': ['SAMEORIGIN'], + 'X-Moose': ['majestic'], + 'X-Served-By': ['cache-cdg8730-CDG'], + 'X-Timer': ['S1481214079.394283,VS0,VE159'], + 'X-Ua-Compatible': ['IE=edge'], + 'X-Xss-Protection': ['1; mode=block']} .. _topics-shell-inspect-response: @@ -263,16 +262,16 @@ When you run the spider, you will get something similar to this:: >>> response.url 'http://example.org' -Then, you can check if the extraction code is working:: +Then, you can check if the extraction code is working: - >>> response.xpath('//h1[@class="fn"]') - [] +>>> response.xpath('//h1[@class="fn"]') +[] Nope, it doesn't. So you can open the response in your web browser and see if -it's the response you were expecting:: +it's the response you were expecting: - >>> view(response) - True +>>> view(response) +True Finally you hit Ctrl-D (or Ctrl-Z in Windows) to exit the shell and resume the crawling:: diff --git a/docs/topics/stats.rst b/docs/topics/stats.rst index 38648ec55..3dd829ebe 100644 --- a/docs/topics/stats.rst +++ b/docs/topics/stats.rst @@ -57,15 +57,15 @@ Set stat value only if lower than previous:: stats.min_value('min_free_memory_percent', value) -Get stat value:: +Get stat value: - >>> stats.get_value('custom_count') - 1 +>>> stats.get_value('custom_count') +1 -Get all stats:: +Get all stats: - >>> stats.get_stats() - {'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)} +>>> stats.get_stats() +{'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)} Available Stats Collectors ========================== From 3560123090c1660fcfad7c5e7da08c9af503940f Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 5 Dec 2019 19:06:51 +0500 Subject: [PATCH 065/313] Rename ASYNCIO_SUPPORT to ASYNCIO_ENABLED. --- docs/topics/settings.rst | 4 ++-- scrapy/commands/crawl.py | 2 +- scrapy/commands/runspider.py | 2 +- scrapy/settings/default_settings.py | 2 +- scrapy/utils/log.py | 4 ++-- tests/test_commands.py | 8 ++++---- tests/test_crawler.py | 10 +++++----- 7 files changed, 16 insertions(+), 16 deletions(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 43f59f7cc..5cbf7450e 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -160,9 +160,9 @@ to any particular component. In that case the module of that component will be shown, typically an extension, middleware or pipeline. It also means that the component must be enabled in order for the setting to have any effect. -.. setting:: ASYNCIO_SUPPORT +.. setting:: ASYNCIO_ENABLED -ASYNCIO_SUPPORT +ASYNCIO_ENABLED --------------- Default: ``False`` diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index e2e69be49..b50761e4a 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -27,7 +27,7 @@ class Command(ScrapyCommand): def process_options(self, args, opts): ScrapyCommand.process_options(self, args, opts) - if self.settings.getbool('ASYNCIO_SUPPORT'): + if self.settings.getbool('ASYNCIO_ENABLED'): install_asyncio_reactor() try: opts.spargs = arglist_to_dict(opts.spargs) diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index ebd4eb620..bfe844eb5 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -51,7 +51,7 @@ class Command(ScrapyCommand): def process_options(self, args, opts): ScrapyCommand.process_options(self, args, opts) - if self.settings.getbool('ASYNCIO_SUPPORT'): + if self.settings.getbool('ASYNCIO_ENABLED'): install_asyncio_reactor() try: opts.spargs = arglist_to_dict(opts.spargs) diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index c9097bd1f..153b8037a 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -19,7 +19,7 @@ from os.path import join, abspath, dirname AJAXCRAWL_ENABLED = False -ASYNCIO_SUPPORT = False +ASYNCIO_ENABLED = False AUTOTHROTTLE_ENABLED = False AUTOTHROTTLE_DEBUG = False diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index 8c56cfa42..0fe3d1549 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -149,11 +149,11 @@ def log_scrapy_info(settings): {'versions': ", ".join("%s %s" % (name, version) for name, version in scrapy_components_versions() if name != "Scrapy")}) - if settings.getbool('ASYNCIO_SUPPORT'): + if settings.getbool('ASYNCIO_ENABLED'): if is_asyncio_reactor_installed(): logger.debug("Asyncio support enabled") else: - logger.error("ASYNCIO_SUPPORT is on but the Twisted asyncio " + logger.error("ASYNCIO_ENABLED is on but the Twisted asyncio " "reactor is not installed, this is not supported " "and asyncio coroutines will not work.") diff --git a/tests/test_commands.py b/tests/test_commands.py index 3b64bfa23..197d80217 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -295,12 +295,12 @@ class BadSpider(scrapy.Spider): self.assertIn("start_requests", log) self.assertIn("badspider.py", log) - def test_asyncio_support_true(self): - log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_SUPPORT=True']) + def test_asyncio_enabled_true(self): + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_ENABLED=True']) self.assertIn("DEBUG: Asyncio support enabled", log) - def test_asyncio_support_false(self): - log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_SUPPORT=False']) + def test_asyncio_enabled_false(self): + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_ENABLED=False']) self.assertNotIn("DEBUG: Asyncio support enabled", log) diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 9410b0e7a..05909d995 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -262,19 +262,19 @@ class CrawlerRunnerHasSpider(unittest.TestCase): self.assertEqual(runner.bootstrap_failed, True) @defer.inlineCallbacks - def test_crawler_process_asyncio_supported_true(self): + def test_crawler_process_asyncio_enabled_true(self): with LogCapture(level=logging.DEBUG) as log: - runner = CrawlerProcess(settings={'ASYNCIO_SUPPORT': True}) + runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': True}) yield runner.crawl(NoRequestsSpider) if self.reactor_pytest == 'asyncio': self.assertIn("Asyncio support enabled", str(log)) else: self.assertNotIn("Asyncio support enabled", str(log)) - self.assertIn("ASYNCIO_SUPPORT is on but the Twisted asyncio reactor is not installed", str(log)) + self.assertIn("ASYNCIO_ENABLED is on but the Twisted asyncio reactor is not installed", str(log)) @defer.inlineCallbacks - def test_crawler_process_asyncio_supported_false(self): - runner = CrawlerProcess(settings={'ASYNCIO_SUPPORT': False}) + def test_crawler_process_asyncio_enabled_false(self): + runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': False}) with LogCapture(level=logging.DEBUG) as log: yield runner.crawl(NoRequestsSpider) self.assertNotIn("Asyncio support enabled", str(log)) From 69cd2e247efe1823ab188eacb241b9e5596a879c Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Sat, 7 Dec 2019 00:09:53 +0500 Subject: [PATCH 066/313] Move a bunch of "from twisted.internet import reactor" inside functions. --- scrapy/cmdline.py | 4 +--- scrapy/commands/shell.py | 3 +-- scrapy/crawler.py | 7 ++++++- scrapy/shell.py | 3 ++- scrapy/utils/defer.py | 4 +++- scrapy/utils/ossignal.py | 3 +-- scrapy/utils/reactor.py | 4 +++- 7 files changed, 17 insertions(+), 11 deletions(-) diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index 3c2efe58f..69e917004 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -6,6 +6,7 @@ import inspect import pkg_resources import scrapy +from scrapy.crawler import CrawlerProcess from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError from scrapy.utils.misc import walk_modules @@ -140,9 +141,6 @@ def execute(argv=None, settings=None): opts, args = parser.parse_args(args=argv[1:]) _run_print_help(parser, cmd.process_options, args, opts) - # needs to be after cmd.process_options() as it imports twisted.internet.reactor - # while commands may want to install the asyncio reactor - from scrapy.crawler import CrawlerProcess cmd.crawler_process = CrawlerProcess(settings) _run_print_help(parser, _run_command, cmd, args, opts) sys.exit(cmd.exitcode) diff --git a/scrapy/commands/shell.py b/scrapy/commands/shell.py index 7516e2aba..d44a32d5f 100644 --- a/scrapy/commands/shell.py +++ b/scrapy/commands/shell.py @@ -7,6 +7,7 @@ from threading import Thread from scrapy.commands import ScrapyCommand from scrapy.http import Request +from scrapy.shell import Shell from scrapy.utils.spider import spidercls_for_request, DefaultSpider from scrapy.utils.url import guess_scheme @@ -69,8 +70,6 @@ class Command(ScrapyCommand): self._start_crawler_thread() - # moved from the top-level because it imports twisted.internet.reactor - from scrapy.shell import Shell shell = Shell(crawler, update_vars=self.update_vars, code=opts.code) shell.start(url=url, redirect=not opts.no_redirect) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 6c7eb737b..450260004 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -3,7 +3,7 @@ import pprint import signal import warnings -from twisted.internet import reactor, defer +from twisted.internet import defer from zope.interface.verify import verifyClass, DoesNotImplement from scrapy import Spider @@ -261,6 +261,7 @@ class CrawlerProcess(CrawlerRunner): log_scrapy_info(self.settings) def _signal_shutdown(self, signum, _): + from twisted.internet import reactor install_shutdown_handlers(self._signal_kill) signame = signal_names[signum] logger.info("Received %(signame)s, shutting down gracefully. Send again to force ", @@ -268,6 +269,7 @@ class CrawlerProcess(CrawlerRunner): reactor.callFromThread(self._graceful_stop_reactor) def _signal_kill(self, signum, _): + from twisted.internet import reactor install_shutdown_handlers(signal.SIG_IGN) signame = signal_names[signum] logger.info('Received %(signame)s twice, forcing unclean shutdown', @@ -286,6 +288,7 @@ class CrawlerProcess(CrawlerRunner): :param boolean stop_after_crawl: stop or not the reactor when all crawlers have finished """ + from twisted.internet import reactor if stop_after_crawl: d = self.join() # Don't start the reactor if the deferreds are already fired @@ -300,6 +303,7 @@ class CrawlerProcess(CrawlerRunner): reactor.run(installSignalHandlers=False) # blocking call def _get_dns_resolver(self): + from twisted.internet import reactor if self.settings.getbool('DNSCACHE_ENABLED'): cache_size = self.settings.getint('DNSCACHE_SIZE') else: @@ -316,6 +320,7 @@ class CrawlerProcess(CrawlerRunner): return d def _stop_reactor(self, _=None): + from twisted.internet import reactor try: reactor.stop() except RuntimeError: # raised if already stopped or in shutdown stage diff --git a/scrapy/shell.py b/scrapy/shell.py index a649d555f..a23b04df9 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -7,7 +7,7 @@ import os import signal import warnings -from twisted.internet import reactor, threads, defer +from twisted.internet import threads, defer from twisted.python import threadable from w3lib.url import any_to_uri @@ -98,6 +98,7 @@ class Shell(object): return spider def fetch(self, request_or_url, spider=None, redirect=True, **kwargs): + from twisted.internet import reactor if isinstance(request_or_url, Request): request = request_or_url else: diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 3b7ef75ab..6a91776c7 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -5,7 +5,7 @@ import asyncio import asyncio.futures import inspect -from twisted.internet import defer, reactor, task +from twisted.internet import defer, task from twisted.python import failure from scrapy.exceptions import IgnoreRequest @@ -19,6 +19,7 @@ def defer_fail(_failure): It delays by 100ms so reactor has a chance to go through readers and writers before attending pending delayed calls, so do not set delay to zero. """ + from twisted.internet import reactor d = defer.Deferred() reactor.callLater(0.1, d.errback, _failure) return d @@ -31,6 +32,7 @@ def defer_succeed(result): It delays by 100ms so reactor has a chance to go trough readers and writers before attending pending delayed calls, so do not set delay to zero. """ + from twisted.internet import reactor d = defer.Deferred() reactor.callLater(0.1, d.callback, result) return d diff --git a/scrapy/utils/ossignal.py b/scrapy/utils/ossignal.py index 7a7aec9be..45c9cef0c 100644 --- a/scrapy/utils/ossignal.py +++ b/scrapy/utils/ossignal.py @@ -1,7 +1,5 @@ import signal -from twisted.internet import reactor - signal_names = {} for signame in dir(signal): @@ -17,6 +15,7 @@ def install_shutdown_handlers(function, override_sigint=True): SIGINT handler won't be install if there is already a handler in place (e.g. Pdb) """ + from twisted.internet import reactor reactor._handleSignals() signal.signal(signal.SIGTERM, function) if signal.getsignal(signal.SIGINT) == signal.default_int_handler or \ diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 493d26d4c..b98fff6ec 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -1,8 +1,9 @@ -from twisted.internet import reactor, error +from twisted.internet import error def listen_tcp(portrange, host, factory): """Like reactor.listenTCP but tries different ports in a range.""" + from twisted.internet import reactor assert len(portrange) <= 2, "invalid portrange: %s" % portrange if not portrange: return reactor.listenTCP(0, factory, interface=host) @@ -30,6 +31,7 @@ class CallLaterOnce(object): self._call = None def schedule(self, delay=0): + from twisted.internet import reactor if self._call is None: self._call = reactor.callLater(delay, self) From 855bbebc8bb862aa02e48f65fd861b1ddf78b57a Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 13 Dec 2019 18:11:49 +0500 Subject: [PATCH 067/313] Move install_asyncio_reactor() from commands to CrawlerProcess. --- scrapy/commands/crawl.py | 3 --- scrapy/commands/runspider.py | 3 --- scrapy/crawler.py | 3 +++ 3 files changed, 3 insertions(+), 6 deletions(-) diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index b50761e4a..8093fd402 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -1,6 +1,5 @@ import os from scrapy.commands import ScrapyCommand -from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.conf import arglist_to_dict from scrapy.utils.python import without_none_values from scrapy.exceptions import UsageError @@ -27,8 +26,6 @@ class Command(ScrapyCommand): def process_options(self, args, opts): ScrapyCommand.process_options(self, args, opts) - if self.settings.getbool('ASYNCIO_ENABLED'): - install_asyncio_reactor() try: opts.spargs = arglist_to_dict(opts.spargs) except ValueError: diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index bfe844eb5..57d8471ca 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -2,7 +2,6 @@ import sys import os from importlib import import_module -from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.spider import iter_spider_classes from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError @@ -51,8 +50,6 @@ class Command(ScrapyCommand): def process_options(self, args, opts): ScrapyCommand.process_options(self, args, opts) - if self.settings.getbool('ASYNCIO_ENABLED'): - install_asyncio_reactor() try: opts.spargs = arglist_to_dict(opts.spargs) except ValueError: diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 450260004..706c8a59d 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -14,6 +14,7 @@ from scrapy.extension import ExtensionManager from scrapy.settings import overridden_settings, Settings from scrapy.signalmanager import SignalManager from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.asyncio import install_asyncio_reactor from scrapy.utils.ossignal import install_shutdown_handlers, signal_names from scrapy.utils.misc import load_object from scrapy.utils.log import ( @@ -256,6 +257,8 @@ class CrawlerProcess(CrawlerRunner): def __init__(self, settings=None, install_root_handler=True): super(CrawlerProcess, self).__init__(settings) + if self.settings.getbool('ASYNCIO_ENABLED'): + install_asyncio_reactor() install_shutdown_handlers(self._signal_shutdown) configure_logging(self.settings, install_root_handler) log_scrapy_info(self.settings) From bfb78b8dea44a5db3f4a3bca83ab58c7ca0e3ef3 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 13 Dec 2019 18:12:07 +0500 Subject: [PATCH 068/313] Add CrawlerProcess tests for ASYNCIO_ENABLED. --- .../asyncio_enabled_no_reactor.py | 17 ++++++++++++++ .../CrawlerProcess/asyncio_enabled_reactor.py | 22 +++++++++++++++++++ tests/test_crawler.py | 11 ++++++++++ 3 files changed, 50 insertions(+) create mode 100644 tests/CrawlerProcess/asyncio_enabled_no_reactor.py create mode 100644 tests/CrawlerProcess/asyncio_enabled_reactor.py diff --git a/tests/CrawlerProcess/asyncio_enabled_no_reactor.py b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py new file mode 100644 index 000000000..dfe028ef4 --- /dev/null +++ b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py @@ -0,0 +1,17 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class NoRequestsSpider(scrapy.Spider): + name = 'no_request' + + def start_requests(self): + return [] + + +process = CrawlerProcess(settings={ + 'ASYNCIO_ENABLED': True, +}) + +process.crawl(NoRequestsSpider) +process.start() diff --git a/tests/CrawlerProcess/asyncio_enabled_reactor.py b/tests/CrawlerProcess/asyncio_enabled_reactor.py new file mode 100644 index 000000000..7a172ea28 --- /dev/null +++ b/tests/CrawlerProcess/asyncio_enabled_reactor.py @@ -0,0 +1,22 @@ +import asyncio + +from twisted.internet import asyncioreactor +asyncioreactor.install(asyncio.get_event_loop()) + +import scrapy +from scrapy.crawler import CrawlerProcess + + +class NoRequestsSpider(scrapy.Spider): + name = 'no_request' + + def start_requests(self): + return [] + + +process = CrawlerProcess(settings={ + 'ASYNCIO_ENABLED': True, +}) + +process.crawl(NoRequestsSpider) +process.start() diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 05909d995..0b2645280 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -301,3 +301,14 @@ class CrawlerProcessSubprocess(unittest.TestCase): def test_simple(self): log = self.run_script('simple.py') self.assertIn('Spider closed (finished)', log) + self.assertNotIn("DEBUG: Asyncio support enabled", log) + + def test_asyncio_enabled_no_reactor(self): + log = self.run_script('asyncio_enabled_no_reactor.py') + self.assertIn('Spider closed (finished)', log) + self.assertIn("DEBUG: Asyncio support enabled", log) + + def test_asyncio_enabled_reactor(self): + log = self.run_script('asyncio_enabled_reactor.py') + self.assertIn('Spider closed (finished)', log) + self.assertIn("DEBUG: Asyncio support enabled", log) From a4ef9750f9058fbe1041bad0fb28f39c693e5659 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Fri, 13 Dec 2019 14:32:06 +0100 Subject: [PATCH 069/313] Fix Flake8-reported issues --- pytest.ini | 2 +- tests/test_linkextractors.py | 12 +++++++----- 2 files changed, 8 insertions(+), 6 deletions(-) diff --git a/pytest.ini b/pytest.ini index 33c34b8e8..02014d1a5 100644 --- a/pytest.ini +++ b/pytest.ini @@ -88,7 +88,7 @@ flake8-ignore = scrapy/http/response/__init__.py E501 E128 W293 W291 scrapy/http/response/text.py E501 W293 E128 E124 # scrapy/linkextractors - scrapy/linkextractors/__init__.py E731 E502 E501 E402 + scrapy/linkextractors/__init__.py E731 E502 E501 E402 W504 scrapy/linkextractors/lxmlhtml.py E501 E731 E226 # scrapy/loader scrapy/loader/__init__.py E501 E502 E128 diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index ebe497913..0ffeaecc3 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -514,25 +514,27 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): """Make sure the FilteringLinkExtractor deprecation warning is not issued for LxmlLinkExtractor""" with catch_warnings(record=True) as warnings: - extractor = LxmlLinkExtractor() + LxmlLinkExtractor() self.assertEqual(len(warnings), 0) + class SubclassedItem(LxmlLinkExtractor): pass - subclassed_extractor = SubclassedItem() + + SubclassedItem() self.assertEqual(len(warnings), 0) class FilteringLinkExtractorTest(unittest.TestCase): def test_deprecation_warning(self): - args = [None]*10 + args = [None] * 10 with catch_warnings(record=True) as warnings: - extractor = FilteringLinkExtractor(*args) + FilteringLinkExtractor(*args) self.assertEqual(len(warnings), 1) self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) with catch_warnings(record=True) as warnings: class SubclassedFilteringLinkExtractor(FilteringLinkExtractor): pass - subclassed_extractor = SubclassedFilteringLinkExtractor(*args) + SubclassedFilteringLinkExtractor(*args) self.assertEqual(len(warnings), 1) self.assertEqual(warnings[0].category, ScrapyDeprecationWarning) From afc886e57865e82e63f3f8f3326f481908919086 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 13 Dec 2019 19:34:47 +0500 Subject: [PATCH 070/313] Simplify tox.ini asyncio entries. --- tox.ini | 10 ++++++---- 1 file changed, 6 insertions(+), 4 deletions(-) diff --git a/tox.ini b/tox.ini index a4edae439..795c20233 100644 --- a/tox.ini +++ b/tox.ini @@ -107,14 +107,16 @@ deps = reppy robotexclusionrulesparser +[asyncio] +commands = + py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests} + [testenv:py35-asyncio] basepython = python3.5 deps = {[testenv]deps} -commands = - py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests} +commands = {[asyncio]commands} [testenv:py38-asyncio] basepython = python3.8 deps = {[testenv]deps} -commands = - py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests} +commands = {[asyncio]commands} From a1605cade6286dd5f7f1c9e4c9660d44ed15ed19 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 13 Dec 2019 19:35:09 +0500 Subject: [PATCH 071/313] Hide utils.defer.isfuture(). --- scrapy/utils/defer.py | 7 +++---- 1 file changed, 3 insertions(+), 4 deletions(-) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 6a91776c7..530bf0e9d 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -2,7 +2,6 @@ Helper functions for dealing with Twisted deferreds """ import asyncio -import asyncio.futures import inspect from twisted.internet import defer, task @@ -121,18 +120,18 @@ def iter_errback(iterable, errback, *a, **kw): errback(failure.Failure(), *a, **kw) -def isfuture(o): +def _isfuture(o): # workaround for Python before 3.5.3 not having asyncio.isfuture if hasattr(asyncio, 'isfuture'): return asyncio.isfuture(o) - return isinstance(o, asyncio.futures.Future) + return isinstance(o, asyncio.Future) def deferred_from_coro(o, asyncio_enabled=False): """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" if isinstance(o, defer.Deferred): return o - if isfuture(o) or inspect.isawaitable(o): + if _isfuture(o) or inspect.isawaitable(o): if not asyncio_enabled: # wrapping the coroutine directly into a Deferred, this doesn't work correctly with coroutines # that use asyncio, e.g. "await asyncio.sleep(1)" From 2db7d453788f5c638d0921b0f7f8bab58e2a58bc Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 19:24:25 +0500 Subject: [PATCH 072/313] Enable skipping tests based on --reactor. --- conftest.py | 8 ++++++++ pytest.ini | 2 ++ 2 files changed, 10 insertions(+) diff --git a/conftest.py b/conftest.py index 64136b48d..56d552953 100644 --- a/conftest.py +++ b/conftest.py @@ -35,6 +35,14 @@ def pytest_collection_modifyitems(session, config, items): except ImportError: pass + @pytest.fixture() def reactor_pytest(request): request.cls.reactor_pytest = request.config.getoption("--reactor") + return request.cls.reactor_pytest + + +@pytest.fixture(autouse=True) +def only_asyncio(request, reactor_pytest): + if request.node.get_closest_marker('only_asyncio') and reactor_pytest != 'asyncio': + pytest.skip('This test is only run with --reactor-asyncio') diff --git a/pytest.ini b/pytest.ini index 336ef041d..7b62a1bd8 100644 --- a/pytest.ini +++ b/pytest.ini @@ -19,6 +19,8 @@ addopts = --ignore=docs/topics/telnetconsole.rst --ignore=docs/utils twisted = 1 +markers = + only_asyncio: marks tests as only enabled when --reactor=asyncio is passed flake8-ignore = # Files that are only meant to provide top-level imports are expected not # to use any of their imports: From 451e7a616e6d23e88528836296a9abdc887834c6 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 12 Jul 2019 00:26:01 -0300 Subject: [PATCH 073/313] Scan callbacks/errbacks for return statements with values different than None --- scrapy/core/scraper.py | 19 ++++-- scrapy/utils/datatypes.py | 31 ++++++++- scrapy/utils/misc.py | 46 +++++++++++++ tests/test_utils_datatypes.py | 66 ++++++++++++++++++- ...t_return_with_argument_inside_generator.py | 37 +++++++++++ 5 files changed, 190 insertions(+), 9 deletions(-) create mode 100644 tests/test_utils_misc/test_return_with_argument_inside_generator.py diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index b3d585cce..99114d3bb 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -9,7 +9,7 @@ from twisted.internet import defer from scrapy.utils.defer import defer_result, defer_succeed, parallel, iter_errback from scrapy.utils.spider import iterate_spider_output -from scrapy.utils.misc import load_object +from scrapy.utils.misc import load_object, warn_on_generator_with_return_value from scrapy.utils.log import logformatter_adapter, failure_to_exc_info from scrapy.exceptions import CloseSpider, DropItem, IgnoreRequest from scrapy import signals @@ -18,6 +18,7 @@ from scrapy.item import BaseItem from scrapy.core.spidermw import SpiderMiddlewareManager from scrapy.utils.request import referer_str + logger = logging.getLogger(__name__) @@ -99,11 +100,13 @@ class Scraper(object): def enqueue_scrape(self, response, request, spider): slot = self.slot dfd = slot.add_response_request(response, request) + def finish_scraping(_): slot.finish_response(response, request) self._check_if_closing(spider, slot) self._scrape_next(spider, slot) return _ + dfd.addBoth(finish_scraping) dfd.addErrback( lambda f: logger.error('Scraper bug processing %(request)s', @@ -123,7 +126,7 @@ class Scraper(object): callback/errback""" assert isinstance(response, (Response, Failure)) - dfd = self._scrape2(response, request, spider) # returns spiders processed output + dfd = self._scrape2(response, request, spider) # returns spider's processed output dfd.addErrback(self.handle_spider_error, request, response, spider) dfd.addCallback(self.handle_spider_output, request, response, spider) return dfd @@ -142,7 +145,10 @@ class Scraper(object): def call_spider(self, result, request, spider): result.request = request dfd = defer_result(result) - dfd.addCallbacks(callback=request.callback or spider.parse, + callback = request.callback or spider.parse + warn_on_generator_with_return_value(spider, callback) + warn_on_generator_with_return_value(spider, request.errback) + dfd.addCallbacks(callback=callback, errback=request.errback, callbackKeywords=request.cb_kwargs) return dfd.addCallback(iterate_spider_output) @@ -172,8 +178,8 @@ class Scraper(object): if not result: return defer_succeed(None) it = iter_errback(result, self.handle_spider_error, request, response, spider) - dfd = parallel(it, self.concurrent_items, - self._process_spidermw_output, request, response, spider) + dfd = parallel(it, self.concurrent_items, self._process_spidermw_output, + request, response, spider) return dfd def _process_spidermw_output(self, output, request, response, spider): @@ -200,8 +206,7 @@ class Scraper(object): """Log and silence errors that come from the engine (typically download errors that got propagated thru here) """ - if (isinstance(download_failure, Failure) and - not download_failure.check(IgnoreRequest)): + if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest): if download_failure.frames: logger.error('Error downloading %(request)s', {'request': request}, diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index ffd1537c3..a52bbc70e 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -8,6 +8,7 @@ This module must not depend on any module outside the Standard Library. import collections import copy import warnings +import weakref from collections.abc import Mapping from scrapy.exceptions import ScrapyDeprecationWarning @@ -240,7 +241,6 @@ class LocalCache(collections.OrderedDict): """Dictionary with a finite number of keys. Older items expires first. - """ def __init__(self, limit=None): @@ -254,6 +254,35 @@ class LocalCache(collections.OrderedDict): super(LocalCache, self).__setitem__(key, value) +class LocalWeakReferencedCache(weakref.WeakKeyDictionary): + """ + A weakref.WeakKeyDictionary implementation that uses LocalCache as its + underlying data structure, making it ordered and capable of being size-limited. + + Useful for memoization, while avoiding keeping received + arguments in memory only because of the cached references. + + Note: like LocalCache and unlike weakref.WeakKeyDictionary, + it cannot be instantiated with an initial dictionary. + """ + + def __init__(self, limit=None): + super(LocalWeakReferencedCache, self).__init__() + self.data = LocalCache(limit=limit) + + def __setitem__(self, key, value): + try: + super(LocalWeakReferencedCache, self).__setitem__(key, value) + except TypeError: + pass # key is not weak-referenceable, skip caching + + def __getitem__(self, key): + try: + return super(LocalWeakReferencedCache, self).__getitem__(key) + except TypeError: + return None # key is not weak-referenceable, it's not cached + + class SequenceExclude(object): """Object to test if an item is NOT within some sequence.""" diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index 9955fb1e7..cb0ee5af3 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -1,13 +1,18 @@ """Helper functions which don't fit anywhere else""" +import ast +import inspect import os import re import hashlib +import warnings from contextlib import contextmanager from importlib import import_module from pkgutil import iter_modules +from textwrap import dedent from w3lib.html import replace_entities +from scrapy.utils.datatypes import LocalWeakReferencedCache from scrapy.utils.python import flatten, to_unicode from scrapy.item import BaseItem @@ -161,3 +166,44 @@ def set_environ(**kwargs): del os.environ[k] else: os.environ[k] = v + + +_generator_callbacks_cache = LocalWeakReferencedCache(limit=128) + + +def is_generator_with_return_value(callable): + """ + Returns True if a callable is a generator function which includes a + 'return' statement with a value different than None, False otherwise + """ + if callable in _generator_callbacks_cache: + return _generator_callbacks_cache[callable] + + def returns_none(return_node): + value = return_node.value + return value is None or isinstance(value, ast.NameConstant) and value.value is None + + if inspect.isgeneratorfunction(callable): + tree = ast.parse(dedent(inspect.getsource(callable))) + for node in ast.walk(tree): + if isinstance(node, ast.Return) and not returns_none(node): + _generator_callbacks_cache[callable] = True + return _generator_callbacks_cache[callable] + + _generator_callbacks_cache[callable] = False + return _generator_callbacks_cache[callable] + + +def warn_on_generator_with_return_value(spider, callable): + """ + Logs a warning if a callable is a generator function and includes + a 'return' statement with a value different than None + """ + if is_generator_with_return_value(callable): + warnings.warn( + 'The "{}.{}" method is a generator and includes a "return" statement with a ' + 'value different than None. This could lead to unexpected behaviour. Please see ' + 'https://docs.python.org/3/reference/simple_stmts.html#the-return-statement ' + 'for details about the semantics of the "return" statement within generators' + .format(spider.__class__.__name__, callable.__name__), stacklevel=2, + ) diff --git a/tests/test_utils_datatypes.py b/tests/test_utils_datatypes.py index 38a25778e..e5aa56eb9 100644 --- a/tests/test_utils_datatypes.py +++ b/tests/test_utils_datatypes.py @@ -2,7 +2,9 @@ import copy import unittest from collections.abc import Mapping, MutableMapping -from scrapy.utils.datatypes import CaselessDict, LocalCache, SequenceExclude +from scrapy.http import Request +from scrapy.utils.datatypes import CaselessDict, LocalCache, LocalWeakReferencedCache, SequenceExclude +from scrapy.utils.python import garbage_collect __doctests__ = ['scrapy.utils.datatypes'] @@ -255,5 +257,67 @@ class LocalCacheTest(unittest.TestCase): self.assertEqual(cache[str(x)], x) +class LocalWeakReferencedCacheTest(unittest.TestCase): + + def test_cache_with_limit(self): + cache = LocalWeakReferencedCache(limit=2) + r1 = Request('https://example.org') + r2 = Request('https://example.com') + r3 = Request('https://example.net') + cache[r1] = 1 + cache[r2] = 2 + cache[r3] = 3 + self.assertEqual(len(cache), 2) + self.assertNotIn(r1, cache) + self.assertIn(r2, cache) + self.assertIn(r3, cache) + self.assertEqual(cache[r2], 2) + self.assertEqual(cache[r3], 3) + del r2 + + # PyPy takes longer to collect dead references + garbage_collect() + + self.assertEqual(len(cache), 1) + + def test_cache_non_weak_referenceable_objects(self): + cache = LocalWeakReferencedCache() + k1 = None + k2 = 1 + k3 = [1, 2, 3] + cache[k1] = 1 + cache[k2] = 2 + cache[k3] = 3 + self.assertNotIn(k1, cache) + self.assertNotIn(k2, cache) + self.assertNotIn(k3, cache) + self.assertEqual(len(cache), 0) + + def test_cache_without_limit(self): + max = 10**4 + cache = LocalWeakReferencedCache() + refs = [] + for x in range(max): + refs.append(Request('https://example.org/{}'.format(x))) + cache[refs[-1]] = x + self.assertEqual(len(cache), max) + for i, r in enumerate(refs): + self.assertIn(r, cache) + self.assertEqual(cache[r], i) + del r # delete reference to the last object in the list + + # delete half of the objects, make sure that is reflected in the cache + for _ in range(max // 2): + refs.pop() + + # PyPy takes longer to collect dead references + garbage_collect() + + self.assertEqual(len(cache), max // 2) + for i, r in enumerate(refs): + self.assertIn(r, cache) + self.assertEqual(cache[r], i) + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_misc/test_return_with_argument_inside_generator.py b/tests/test_utils_misc/test_return_with_argument_inside_generator.py new file mode 100644 index 000000000..bdbec1beb --- /dev/null +++ b/tests/test_utils_misc/test_return_with_argument_inside_generator.py @@ -0,0 +1,37 @@ +import unittest + +from scrapy.utils.misc import is_generator_with_return_value + + +class UtilsMiscPy3TestCase(unittest.TestCase): + + def test_generators_with_return_statements(self): + def f(): + yield 1 + return 2 + + def g(): + yield 1 + return 'asdf' + + def h(): + yield 1 + return None + + def i(): + yield 1 + return + + def j(): + yield 1 + + def k(): + yield 1 + yield from g() + + assert is_generator_with_return_value(f) + assert is_generator_with_return_value(g) + assert not is_generator_with_return_value(h) + assert not is_generator_with_return_value(i) + assert not is_generator_with_return_value(j) + assert not is_generator_with_return_value(k) # not recursive From 039e6fe6919341dbfd864c4a406d35389c8e2992 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 20:17:41 +0500 Subject: [PATCH 074/313] Refactor install_asyncio_reactor slightly. --- scrapy/utils/asyncio.py | 16 +++++++++------- 1 file changed, 9 insertions(+), 7 deletions(-) diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py index b5d5f92d9..b53c8a8b0 100644 --- a/scrapy/utils/asyncio.py +++ b/scrapy/utils/asyncio.py @@ -1,3 +1,8 @@ +from contextlib import suppress + +from twisted.internet.error import ReactorAlreadyInstalledError + + def install_asyncio_reactor(): """ Tries to install AsyncioSelectorReactor """ @@ -5,13 +10,10 @@ def install_asyncio_reactor(): import asyncio from twisted.internet import asyncioreactor except ImportError: - pass - else: - from twisted.internet.error import ReactorAlreadyInstalledError - try: - asyncioreactor.install(asyncio.get_event_loop()) - except ReactorAlreadyInstalledError: - pass + return + + with suppress(ReactorAlreadyInstalledError): + asyncioreactor.install(asyncio.get_event_loop()) def is_asyncio_reactor_installed(): From 900de7c14607fbe2936fa682d03747916337f075 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 21:11:58 +0500 Subject: [PATCH 075/313] Fix the reactor_pytest fixture. --- conftest.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/conftest.py b/conftest.py index 56d552953..6d9696a3f 100644 --- a/conftest.py +++ b/conftest.py @@ -36,8 +36,11 @@ def pytest_collection_modifyitems(session, config, items): pass -@pytest.fixture() +@pytest.fixture(scope='class') def reactor_pytest(request): + if not request.cls: + # doctests + return request.cls.reactor_pytest = request.config.getoption("--reactor") return request.cls.reactor_pytest From 5980d3bbff67d060a1a1b15372293ced972dbe8b Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 10 Sep 2019 14:23:11 +0500 Subject: [PATCH 076/313] Add simple tests for pipelines. --- tests/test_pipelines.py | 71 +++++++++++++++++++++++++++++++++++++++++ 1 file changed, 71 insertions(+) create mode 100644 tests/test_pipelines.py diff --git a/tests/test_pipelines.py b/tests/test_pipelines.py new file mode 100644 index 000000000..bc53f5427 --- /dev/null +++ b/tests/test_pipelines.py @@ -0,0 +1,71 @@ +from twisted.internet import defer +from twisted.internet.defer import Deferred +from twisted.trial import unittest + +from scrapy import Spider, signals, Request +from scrapy.utils.test import get_crawler + +from tests.mockserver import MockServer + + +class SimplePipeline: + def process_item(self, item, spider): + item['pipeline_passed'] = True + return item + + +class DeferredPipeline: + def cb(self, item): + item['pipeline_passed'] = True + return item + + def process_item(self, item, spider): + d = Deferred() + d.addCallback(self.cb) + d.callback(item) + return d + + +class ItemSpider(Spider): + name = 'itemspider' + + def start_requests(self): + yield Request(self.mockserver.url('/status?n=200')) + + def parse(self, response): + return {'field': 42} + + +class PipelineTestCase(unittest.TestCase): + def setUp(self): + self.mockserver = MockServer() + self.mockserver.__enter__() + + def tearDown(self): + self.mockserver.__exit__(None, None, None) + + def _on_item_scraped(self, item): + self.assertIsInstance(item, dict) + self.assertTrue(item.get('pipeline_passed')) + self.items.append(item) + + def _create_crawler(self, pipeline_class): + settings = { + 'ITEM_PIPELINES': {__name__ + '.' + pipeline_class.__name__: 1}, + } + crawler = get_crawler(ItemSpider, settings) + crawler.signals.connect(self._on_item_scraped, signals.item_scraped) + self.items = [] + return crawler + + @defer.inlineCallbacks + def test_simple_pipeline(self): + crawler = self._create_crawler(SimplePipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) + + @defer.inlineCallbacks + def test_deferred_pipeline(self): + crawler = self._create_crawler(DeferredPipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) From 7d0096da6e37e508280e3af7388122bcdf0e3dcf Mon Sep 17 00:00:00 2001 From: apu <1173372284@qq.com> Date: Tue, 17 Dec 2019 09:47:01 +0800 Subject: [PATCH 077/313] Fix mail attachs tcmime *** (#4229) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit When the file name consists of alphanumeric characters, it is normal to receive the attachment name. However,However, problems will occur if the file name is changed to Chinese. This has nothing to do with the file type --- scrapy/mail.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scrapy/mail.py b/scrapy/mail.py index 891bb5e09..9655b8114 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -73,8 +73,7 @@ class MailSender(object): part = MIMEBase(*mimetype.split('/')) part.set_payload(f.read()) Encoders.encode_base64(part) - part.add_header('Content-Disposition', 'attachment; filename="%s"' \ - % attach_name) + part.add_header('Content-Disposition', 'attachment', filename=attach_name) msg.attach(part) else: msg.set_payload(body) From 2d92a39003a5a0b01b265b87cf722a61f121ca60 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Wed, 18 Dec 2019 12:07:08 +0500 Subject: [PATCH 078/313] Restore test_download_with_proxy_https_noconnect, check for a warning there. --- tests/test_downloader_handlers.py | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 45d4aa952..87ab4c9e5 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -33,7 +33,7 @@ from scrapy.responsetypes import responsetypes from scrapy.settings import Settings from scrapy.utils.test import get_crawler, skip_if_no_boto from scrapy.utils.python import to_bytes -from scrapy.exceptions import NotConfigured +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from tests.mockserver import MockServer, ssl_context_factory, Echo from tests.spiders import SingleRequestSpider @@ -687,6 +687,18 @@ class HttpProxyTestCase(unittest.TestCase): request = Request('http://example.com', meta={'proxy': http_proxy}) return self.download_request(request, Spider('foo')).addCallback(_test) + def test_download_with_proxy_https_noconnect(self): + def _test(response): + self.assertEqual(response.status, 200) + self.assertEqual(response.url, request.url) + self.assertEqual(response.body, b'https://example.com') + + http_proxy = '%s?noconnect' % self.getURL('') + request = Request('https://example.com', meta={'proxy': http_proxy}) + with self.assertWarnsRegex(ScrapyDeprecationWarning, + r'Using HTTPS proxies in the noconnect mode is deprecated'): + return self.download_request(request, Spider('foo')).addCallback(_test) + def test_download_without_proxy(self): def _test(response): self.assertEqual(response.status, 200) From bb2ff13e4c7c80cfc7925f60eadc97dec0d69026 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Wed, 18 Dec 2019 15:39:08 +0500 Subject: [PATCH 079/313] Skip Http10ProxyTestCase.test_download_with_proxy_https_noconnect --- tests/test_downloader_handlers.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 87ab4c9e5..412a9c084 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -712,6 +712,8 @@ class HttpProxyTestCase(unittest.TestCase): class Http10ProxyTestCase(HttpProxyTestCase): download_handler_cls = HTTP10DownloadHandler + def test_download_with_proxy_https_noconnect(self): + raise unittest.SkipTest('noconnect is not supported in HTTP10DownloadHandler') class Http11ProxyTestCase(HttpProxyTestCase): download_handler_cls = HTTP11DownloadHandler From ac302c3f615d339ed36000231ca3e1e1c347478a Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Wed, 18 Dec 2019 15:43:05 +0500 Subject: [PATCH 080/313] Fix a flake8 problem. --- tests/test_downloader_handlers.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 412a9c084..7412d7ebf 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -715,6 +715,7 @@ class Http10ProxyTestCase(HttpProxyTestCase): def test_download_with_proxy_https_noconnect(self): raise unittest.SkipTest('noconnect is not supported in HTTP10DownloadHandler') + class Http11ProxyTestCase(HttpProxyTestCase): download_handler_cls = HTTP11DownloadHandler From 12f9ffeb5d1c1f4d02f616852dcae82ef633e8d9 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 18 Dec 2019 10:53:27 +0000 Subject: [PATCH 081/313] remove requirements-py3.txt --- requirements-py3.txt | 16 ---------------- tox.ini | 1 - 2 files changed, 17 deletions(-) delete mode 100644 requirements-py3.txt diff --git a/requirements-py3.txt b/requirements-py3.txt deleted file mode 100644 index 28c649e28..000000000 --- a/requirements-py3.txt +++ /dev/null @@ -1,16 +0,0 @@ -parsel>=1.5.0 -PyDispatcher>=2.0.5 -Twisted>=17.9.0 -w3lib>=1.17.0 - -pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'" -queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'" -cryptography>=2.0 # Earlier versions would fail to install - -# Reference versions taken from -# https://packages.ubuntu.com/xenial/python/ -# https://packages.ubuntu.com/xenial/zope/ -cssselect>=0.9.1 -lxml>=3.5.0 -service_identity>=16.0.0 -zope.interface>=4.1.3 diff --git a/tox.ini b/tox.ini index fd75d18e2..f37c381d0 100644 --- a/tox.ini +++ b/tox.ini @@ -9,7 +9,6 @@ envlist = py35 [testenv] deps = -ctests/constraints.txt - -rrequirements-py3.txt -rtests/requirements-py3.txt # Extras botocore>=1.3.23 From ee9881d2704798c9cd61b6da503bb0694227c58c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Wed, 18 Dec 2019 12:08:34 +0100 Subject: [PATCH 082/313] Improve FilteringLinkExtractor.__new__ --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index a510fef70..7254bd79c 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -61,7 +61,7 @@ class FilteringLinkExtractor(object): warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' 'please use scrapy.linkextractors.LinkExtractor instead', ScrapyDeprecationWarning, stacklevel=2) - return super(FilteringLinkExtractor, cls).__new__(cls) + return super().__new__(cls, *args, **kwargs) def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text): From 174769a3f08fcd84eaec8a88217a05f8ebc3f2cb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Wed, 18 Dec 2019 12:09:03 +0100 Subject: [PATCH 083/313] Use a better name for the LxmlLinkExtractor subclassing test --- tests/test_linkextractors.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index 0ffeaecc3..cfd4c6b85 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -517,10 +517,10 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): LxmlLinkExtractor() self.assertEqual(len(warnings), 0) - class SubclassedItem(LxmlLinkExtractor): + class SubclassedLxmlLinkExtractor(LxmlLinkExtractor): pass - SubclassedItem() + SubclassedLxmlLinkExtractor() self.assertEqual(len(warnings), 0) From 012533924a799c7bf1d3f2d4a489d8ca1ac462f8 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 18 Dec 2019 11:13:36 +0000 Subject: [PATCH 084/313] remove requirements from here too --- docs/requirements.txt | 1 - 1 file changed, 1 deletion(-) diff --git a/docs/requirements.txt b/docs/requirements.txt index 0ed11c4dc..773b92cea 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1,4 +1,3 @@ --r ../requirements-py3.txt Sphinx>=2.1 sphinx-hoverxref sphinx-notfound-page From 7ccb169a27ccbedf3145b824e1bd655cb6902f32 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Wed, 18 Dec 2019 19:41:16 +0500 Subject: [PATCH 085/313] Split a long test in test_engine.py into three. --- tests/test_engine.py | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/tests/test_engine.py b/tests/test_engine.py index a48b63025..25dee7c1f 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -179,7 +179,6 @@ class EngineTest(unittest.TestCase): @defer.inlineCallbacks def test_crawler(self): - for spider in TestSpider, DictItemsSpider: self.run = CrawlerRun(spider) yield self.run.run() @@ -189,11 +188,15 @@ class EngineTest(unittest.TestCase): self._assert_scraped_items() self._assert_signals_catched() + @defer.inlineCallbacks + def test_crawler_dupefilter(self): self.run = CrawlerRun(TestDupeFilterSpider) yield self.run.run() self._assert_scheduled_requests(urls_to_visit=7) self._assert_dropped_requests() + @defer.inlineCallbacks + def test_crawler_itemerror(self): self.run = CrawlerRun(ItemZeroDivisionErrorSpider) yield self.run.run() self._assert_items_error() From 916382e109dc06a57f4631448555953d1fa540b5 Mon Sep 17 00:00:00 2001 From: elacuesta <elacuesta@users.noreply.github.com> Date: Wed, 18 Dec 2019 12:05:33 -0300 Subject: [PATCH 086/313] Add errback parameter to scrapy.spiders.crawl.Rule (#4000) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add errback parameter to scrapy.spiders.crawl.Rule * CrawlSpider: optimize by reducing iterations * [test] Rule.errback * [doc] Rule.errback * [doc] Use autoclass in docs/topics/spiders.rst Co-Authored-By: Adrián Chaves <adrian@chaves.io> * Rule.process_links takes a list * Fix aesthetic issue reported by Flake8 --- docs/topics/spiders.rst | 6 +++++ scrapy/spiders/crawl.py | 60 ++++++++++++++++++++++++++--------------- tests/mockserver.py | 7 +++++ tests/spiders.py | 36 +++++++++++++++++++++++-- tests/test_crawl.py | 18 ++++++++++--- 5 files changed, 101 insertions(+), 26 deletions(-) diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index d65a43afd..b0fb14e24 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -414,6 +414,12 @@ Crawling rules from which the request originated as second argument. It must return a ``Request`` object or ``None`` (to filter out the request). + ``errback`` is a callable or a string (in which case a method from the spider + object with that name will be used) to be called if any exception is + raised while processing a request generated by the rule. + It receives a :class:`Twisted Failure <twisted.python.failure.Failure>` + instance as first parameter. + CrawlSpider example ~~~~~~~~~~~~~~~~~~~ diff --git a/scrapy/spiders/crawl.py b/scrapy/spiders/crawl.py index a5eb1a518..a2c364c0e 100644 --- a/scrapy/spiders/crawl.py +++ b/scrapy/spiders/crawl.py @@ -16,7 +16,11 @@ from scrapy.utils.python import get_func_args from scrapy.utils.spider import iterate_spider_output -def _identity(request, response): +def _identity(x): + return x + + +def _identity_process_request(request, response): return request @@ -32,17 +36,20 @@ _default_link_extractor = LinkExtractor() class Rule(object): - def __init__(self, link_extractor=None, callback=None, cb_kwargs=None, follow=None, process_links=None, process_request=None): + def __init__(self, link_extractor=None, callback=None, cb_kwargs=None, follow=None, + process_links=None, process_request=None, errback=None): self.link_extractor = link_extractor or _default_link_extractor self.callback = callback + self.errback = errback self.cb_kwargs = cb_kwargs or {} - self.process_links = process_links - self.process_request = process_request or _identity + self.process_links = process_links or _identity + self.process_request = process_request or _identity_process_request self.process_request_argcount = None self.follow = follow if follow is not None else not callback def _compile(self, spider): self.callback = _get_method(self.callback, spider) + self.errback = _get_method(self.errback, spider) self.process_links = _get_method(self.process_links, spider) self.process_request = _get_method(self.process_request, spider) self.process_request_argcount = len(get_func_args(self.process_request)) @@ -76,48 +83,59 @@ class CrawlSpider(Spider): def process_results(self, response, results): return results - def _build_request(self, rule, link): - r = Request(url=link.url, callback=self._response_downloaded) - r.meta.update(rule=rule, link_text=link.text) - return r + def _build_request(self, rule_index, link): + return Request( + url=link.url, + callback=self._callback, + errback=self._errback, + meta=dict(rule=rule_index, link_text=link.text), + ) def _requests_to_follow(self, response): if not isinstance(response, HtmlResponse): return seen = set() - for n, rule in enumerate(self._rules): + for rule_index, rule in enumerate(self._rules): links = [lnk for lnk in rule.link_extractor.extract_links(response) if lnk not in seen] - if links and rule.process_links: - links = rule.process_links(links) - for link in links: + for link in rule.process_links(links): seen.add(link) - request = self._build_request(n, link) + request = self._build_request(rule_index, link) yield rule._process_request(request, response) - def _response_downloaded(self, response): + def _callback(self, response): rule = self._rules[response.meta['rule']] return self._parse_response(response, rule.callback, rule.cb_kwargs, rule.follow) + def _errback(self, failure): + rule = self._rules[failure.request.meta['rule']] + return self._handle_failure(failure, rule.errback) + def _parse_response(self, response, callback, cb_kwargs, follow=True): if callback: cb_res = callback(response, **cb_kwargs) or () cb_res = self.process_results(response, cb_res) - for requests_or_item in iterate_spider_output(cb_res): - yield requests_or_item + for request_or_item in iterate_spider_output(cb_res): + yield request_or_item if follow and self._follow_links: for request_or_item in self._requests_to_follow(response): yield request_or_item + def _handle_failure(self, failure, errback): + if errback: + results = errback(failure) or () + for request_or_item in iterate_spider_output(results): + yield request_or_item + def _compile_rules(self): - self._rules = [copy.copy(r) for r in self.rules] - for rule in self._rules: - rule._compile(self) + self._rules = [] + for rule in self.rules: + self._rules.append(copy.copy(rule)) + self._rules[-1]._compile(self) @classmethod def from_crawler(cls, crawler, *args, **kwargs): spider = super(CrawlSpider, cls).from_crawler(crawler, *args, **kwargs) - spider._follow_links = crawler.settings.getbool( - 'CRAWLSPIDER_FOLLOW_LINKS', True) + spider._follow_links = crawler.settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True) return spider diff --git a/tests/mockserver.py b/tests/mockserver.py index fe28176d4..a45277db9 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -164,6 +164,12 @@ class Drop(Partial): request.finish() +class ArbitraryLengthPayloadResource(LeafResource): + + def render(self, request): + return request.content.read() + + class Root(Resource): def __init__(self): @@ -177,6 +183,7 @@ class Root(Resource): self.putChild(b"echo", Echo()) self.putChild(b"payload", PayloadResource()) self.putChild(b"xpayload", EncodingResourceWrapper(PayloadResource(), [GzipEncoderFactory()])) + self.putChild(b"alpayload", ArbitraryLengthPayloadResource()) try: from tests import tests_datadir self.putChild(b"files", File(os.path.join(tests_datadir, 'test_site/files/'))) diff --git a/tests/spiders.py b/tests/spiders.py index 981bd2eb8..39c8da0b6 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -1,14 +1,14 @@ """ Some spiders used for testing and benchmarking """ - import time from urllib.parse import urlencode -from scrapy.spiders import Spider from scrapy.http import Request from scrapy.item import Item from scrapy.linkextractors import LinkExtractor +from scrapy.spiders import Spider +from scrapy.spiders.crawl import CrawlSpider, Rule class MockServerSpider(Spider): @@ -184,3 +184,35 @@ class DuplicateStartRequestsSpider(MockServerSpider): def parse(self, response): self.visited += 1 + + +class CrawlSpiderWithErrback(MockServerSpider, CrawlSpider): + name = 'crawl_spider_with_errback' + custom_settings = { + 'RETRY_HTTP_CODES': [], # no need to retry + } + rules = ( + Rule(LinkExtractor(), callback='callback', errback='errback', follow=True), + ) + + def start_requests(self): + test_body = b""" + <html> + <head><title>Page title<title></head> + <body> + <p><a href="/status?n=200">Item 200</a></p> <!-- callback --> + <p><a href="/status?n=201">Item 201</a></p> <!-- callback --> + <p><a href="/status?n=404">Item 404</a></p> <!-- errback --> + <p><a href="/status?n=500">Item 500</a></p> <!-- errback --> + <p><a href="/status?n=501">Item 501</a></p> <!-- errback --> + </body> + </html> + """ + url = self.mockserver.url("/alpayload") + yield Request(url, method="POST", body=test_body) + + def callback(self, response): + self.logger.info('[callback] status %i', response.status) + + def errback(self, failure): + self.logger.info('[errback] status %i', failure.value.response.status) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 3307899b7..76f87458b 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -5,12 +5,12 @@ from testfixtures import LogCapture from twisted.internet import defer from twisted.trial.unittest import TestCase -from scrapy.http import Request from scrapy.crawler import CrawlerRunner +from scrapy.http import Request from scrapy.utils.python import to_unicode -from tests.spiders import FollowAllSpider, DelaySpider, SimpleSpider, \ - BrokenStartRequestsSpider, SingleRequestSpider, DuplicateStartRequestsSpider from tests.mockserver import MockServer +from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider, + SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback) class CrawlTestCase(TestCase): @@ -277,3 +277,15 @@ with multiples lines self._assert_retried(log) self.assertIn("Got response 200", str(log)) + + @defer.inlineCallbacks + def test_crawlspider_with_errback(self): + self.runner.crawl(CrawlSpiderWithErrback, mockserver=self.mockserver) + + with LogCapture() as log: + yield self.runner.join() + + self.assertIn("[callback] status 200", str(log)) + self.assertIn("[callback] status 201", str(log)) + self.assertIn("[errback] status 404", str(log)) + self.assertIn("[errback] status 500", str(log)) From a5de2c64e6f1233a98b1b180606fca0ae4dd0871 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Wed, 18 Dec 2019 16:24:48 +0100 Subject: [PATCH 087/313] fix W291, W292, W293 (whitespaces) --- pytest.ini | 27 ++++++++++----------- scrapy/http/response/__init__.py | 4 +-- scrapy/http/response/text.py | 4 +-- scrapy/logformatter.py | 4 +-- scrapy/utils/markup.py | 2 +- scrapy/utils/multipart.py | 2 +- tests/test_contracts.py | 2 +- tests/test_downloadermiddleware_retry.py | 8 +++--- tests/test_dupefilters.py | 6 ++--- tests/test_pipeline_files.py | 1 - tests/test_robotstxt_interface.py | 2 +- tests/test_spidermiddleware_offsite.py | 2 +- tests/test_spidermiddleware_output_chain.py | 8 +++--- 13 files changed, 35 insertions(+), 37 deletions(-) diff --git a/pytest.ini b/pytest.ini index f088e10ef..ac3c8cfb5 100644 --- a/pytest.ini +++ b/pytest.ini @@ -85,8 +85,8 @@ flake8-ignore = scrapy/http/request/__init__.py E501 scrapy/http/request/form.py E501 E123 scrapy/http/request/json_request.py E501 - scrapy/http/response/__init__.py E501 E128 W293 W291 - scrapy/http/response/text.py E501 W293 E128 E124 + scrapy/http/response/__init__.py E501 E128 + scrapy/http/response/text.py E501 E128 E124 # scrapy/linkextractors scrapy/linkextractors/__init__.py E731 E501 E402 scrapy/linkextractors/lxmlhtml.py E501 E731 E226 @@ -127,9 +127,9 @@ flake8-ignore = scrapy/utils/httpobj.py E501 scrapy/utils/iterators.py E501 E701 scrapy/utils/log.py E128 W503 - scrapy/utils/markup.py F403 W292 + scrapy/utils/markup.py F403 scrapy/utils/misc.py E501 E226 - scrapy/utils/multipart.py F403 W292 + scrapy/utils/multipart.py F403 scrapy/utils/project.py E501 scrapy/utils/python.py E501 scrapy/utils/reactor.py E226 @@ -144,7 +144,6 @@ flake8-ignore = scrapy/utils/url.py E501 F403 E128 F405 # scrapy scrapy/__init__.py E402 E501 - scrapy/_monkeypatches.py W293 scrapy/cmdline.py E501 scrapy/crawler.py E501 scrapy/dupefilters.py E501 E202 @@ -153,7 +152,7 @@ flake8-ignore = scrapy/interfaces.py E501 scrapy/item.py E501 E128 scrapy/link.py E501 - scrapy/logformatter.py E501 W293 + scrapy/logformatter.py E501 scrapy/mail.py E402 E128 E501 E502 scrapy/middleware.py E128 E501 scrapy/pqueues.py E501 @@ -174,7 +173,7 @@ flake8-ignore = tests/test_command_parse.py E501 E128 E303 E226 tests/test_command_shell.py E501 E128 tests/test_commands.py E128 E501 - tests/test_contracts.py E501 E128 W293 + tests/test_contracts.py E501 E128 tests/test_crawl.py E501 E741 E265 tests/test_crawler.py F841 E306 E501 tests/test_dependencies.py F841 E501 E305 @@ -189,17 +188,17 @@ flake8-ignore = tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123 tests/test_downloadermiddleware_httpproxy.py E501 E128 tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 - tests/test_downloadermiddleware_retry.py E501 E128 W293 E251 E303 E126 + tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126 tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 - tests/test_dupefilters.py E221 E501 E741 W293 W291 E128 E124 + tests/test_dupefilters.py E221 E501 E741 E128 E124 tests/test_engine.py E401 E501 E128 tests/test_exporters.py E501 E731 E306 E128 E124 tests/test_extension_telnet.py F841 tests/test_feedexport.py E501 F841 E241 tests/test_http_cookies.py E501 tests/test_http_headers.py E501 - tests/test_http_request.py E402 E501 E127 E128 W293 E128 E126 E123 + tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 tests/test_http_response.py E501 E301 E128 E265 tests/test_item.py E701 E128 F841 E306 tests/test_link.py E501 @@ -209,20 +208,20 @@ flake8-ignore = tests/test_mail.py E128 E501 E305 tests/test_middleware.py E501 E128 tests/test_pipeline_crawl.py E131 E501 E128 E126 - tests/test_pipeline_files.py E501 W293 E303 E272 E226 + tests/test_pipeline_files.py E501 E303 E272 E226 tests/test_pipeline_images.py F841 E501 E303 tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502 tests/test_proxy_connect.py E501 E741 tests/test_request_cb_kwargs.py E501 tests/test_responsetypes.py E501 E305 - tests/test_robotstxt_interface.py E501 W291 E501 + tests/test_robotstxt_interface.py E501 E501 tests/test_scheduler.py E501 E126 E123 tests/test_selector.py E501 E127 tests/test_spider.py E501 tests/test_spidermiddleware.py E501 E226 tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 - tests/test_spidermiddleware_offsite.py E501 E128 E111 W293 - tests/test_spidermiddleware_output_chain.py E501 W293 E226 + tests/test_spidermiddleware_offsite.py E501 E128 E111 + tests/test_spidermiddleware_output_chain.py E501 E226 tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 tests/test_squeues.py E501 E701 E741 tests/test_utils_conf.py E501 E303 E128 diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index 64e9c6c20..e79ce9acc 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -113,8 +113,8 @@ class Response(object_ref): It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be a relative URL or a ``scrapy.link.Link`` object, not only an absolute URL. - - :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` + + :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` method which supports selectors in addition to absolute/relative URLs and Link objects. """ diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 1079fd6e8..4f9afde87 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -125,7 +125,7 @@ class TextResponse(Response): Return a :class:`~.Request` instance to follow a link ``url``. It accepts the same arguments as ``Request.__init__`` method, but ``url`` can be not only an absolute URL, but also - + * a relative URL; * a scrapy.link.Link object (e.g. a link extractor result); * an attribute Selector (not SelectorList) - e.g. @@ -133,7 +133,7 @@ class TextResponse(Response): ``response.xpath('//img/@src')[0]``. * a Selector for ``<a>`` or ``<link>`` element, e.g. ``response.css('a.my_link')[0]``. - + See :ref:`response-follow-example` for usage examples. """ if isinstance(url, parsel.Selector): diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 5189d7cfa..4e5963e99 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -13,7 +13,7 @@ ERRORMSG = u"'Error processing %(item)s'" class LogFormatter(object): """Class for generating log messages for different actions. - + All methods must return a dictionary listing the parameters ``level``, ``msg`` and ``args`` which are going to be used for constructing the log message when calling ``logging.log``. @@ -48,7 +48,7 @@ class LogFormatter(object): } } """ - + def crawled(self, request, response, spider): """Logs a message when the crawler finds a webpage.""" request_flags = ' %s' % str(request.flags) if request.flags else '' diff --git a/scrapy/utils/markup.py b/scrapy/utils/markup.py index 2455fcc16..9728c542a 100644 --- a/scrapy/utils/markup.py +++ b/scrapy/utils/markup.py @@ -11,4 +11,4 @@ from w3lib.html import * # noqa: F401 warnings.warn("Module `scrapy.utils.markup` is deprecated. " "Please import from `w3lib.html` instead.", - ScrapyDeprecationWarning, stacklevel=2) \ No newline at end of file + ScrapyDeprecationWarning, stacklevel=2) diff --git a/scrapy/utils/multipart.py b/scrapy/utils/multipart.py index e81f63152..5dcf791b8 100644 --- a/scrapy/utils/multipart.py +++ b/scrapy/utils/multipart.py @@ -12,4 +12,4 @@ from w3lib.form import * # noqa: F401 warnings.warn("Module `scrapy.utils.multipart` is deprecated. " "If you're using `encode_multipart` function, please use " "`urllib3.filepost.encode_multipart_formdata` instead", - ScrapyDeprecationWarning, stacklevel=2) \ No newline at end of file + ScrapyDeprecationWarning, stacklevel=2) diff --git a/tests/test_contracts.py b/tests/test_contracts.py index 582e3d052..11d41c1fe 100644 --- a/tests/test_contracts.py +++ b/tests/test_contracts.py @@ -252,7 +252,7 @@ class ContractsManagerTest(unittest.TestCase): self.assertEqual(len(contracts), 3) self.assertEqual(frozenset(type(x) for x in contracts), frozenset([UrlContract, CallbackKeywordArgumentsContract, ReturnsContract])) - + contracts = self.conman.extract_contracts(spider.returns_item_cb_kwargs) self.assertEqual(len(contracts), 3) self.assertEqual(frozenset(type(x) for x in contracts), diff --git a/tests/test_downloadermiddleware_retry.py b/tests/test_downloadermiddleware_retry.py index e09d66086..9c989977e 100644 --- a/tests/test_downloadermiddleware_retry.py +++ b/tests/test_downloadermiddleware_retry.py @@ -124,7 +124,7 @@ class MaxRetryTimesTest(unittest.TestCase): # SETTINGS: meta(max_retry_times) = 0 meta_max_retry_times = 0 - + req = Request(self.invalid_url, meta={'max_retry_times': meta_max_retry_times}) self._test_retry(req, DNSLookupError('foo'), meta_max_retry_times) @@ -137,7 +137,7 @@ class MaxRetryTimesTest(unittest.TestCase): self._test_retry(req, DNSLookupError('foo'), self.mw.max_retry_times) def test_with_metakey_greater(self): - + # SETINGS: RETRY_TIMES < meta(max_retry_times) self.mw.max_retry_times = 2 meta_max_retry_times = 3 @@ -149,7 +149,7 @@ class MaxRetryTimesTest(unittest.TestCase): self._test_retry(req2, DNSLookupError('foo'), self.mw.max_retry_times) def test_with_metakey_lesser(self): - + # SETINGS: RETRY_TIMES > meta(max_retry_times) self.mw.max_retry_times = 5 meta_max_retry_times = 4 @@ -172,7 +172,7 @@ class MaxRetryTimesTest(unittest.TestCase): self._test_retry(req, DNSLookupError('foo'), 0) def _test_retry(self, req, exception, max_retry_times): - + for i in range(0, max_retry_times): req = self.mw.process_exception(req, exception, self.spider) assert isinstance(req, Request) diff --git a/tests/test_dupefilters.py b/tests/test_dupefilters.py index e4b0bdf83..0546558bc 100644 --- a/tests/test_dupefilters.py +++ b/tests/test_dupefilters.py @@ -142,12 +142,12 @@ class RFPDupeFilterTest(unittest.TestCase): r1 = Request('http://scrapytest.org/index.html') r2 = Request('http://scrapytest.org/index.html') - + dupefilter.log(r1, spider) dupefilter.log(r2, spider) assert crawler.stats.get_value('dupefilter/filtered') == 2 - l.check_present(('scrapy.dupefilters', 'DEBUG', + l.check_present(('scrapy.dupefilters', 'DEBUG', ('Filtered duplicate request: <GET http://scrapytest.org/index.html>' ' - no more duplicates will be shown' ' (see DUPEFILTER_DEBUG to show all duplicates)'))) @@ -169,7 +169,7 @@ class RFPDupeFilterTest(unittest.TestCase): r2 = Request('http://scrapytest.org/index.html', headers={'Referer': 'http://scrapytest.org/INDEX.html'} ) - + dupefilter.log(r1, spider) dupefilter.log(r2, spider) diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index 52f2b554e..141141671 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -58,7 +58,6 @@ class FilesPipelineTestCase(unittest.TestCase): self.assertEqual(file_path(Request("data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAR0AAACxCAMAAADOHZloAAACClBMVEX/\ //+F0tzCwMK76ZKQ21AMqr7oAAC96JvD5aWM2kvZ78J0N7fmAAC46Y4Ap7y")), 'full/178059cbeba2e34120a67f2dc1afc3ecc09b61cb.png') - def test_fs_store(self): assert isinstance(self.pipeline.store, FSFilesStore) diff --git a/tests/test_robotstxt_interface.py b/tests/test_robotstxt_interface.py index 27d79437b..24aaaf7ec 100644 --- a/tests/test_robotstxt_interface.py +++ b/tests/test_robotstxt_interface.py @@ -44,7 +44,7 @@ class BaseRobotParserTest: def test_allowed_wildcards(self): robotstxt_robotstxt_body = """User-agent: first - Disallow: /disallowed/*/end$ + Disallow: /disallowed/*/end$ User-agent: second Allow: /*allowed diff --git a/tests/test_spidermiddleware_offsite.py b/tests/test_spidermiddleware_offsite.py index 992e60be2..7511aa568 100644 --- a/tests/test_spidermiddleware_offsite.py +++ b/tests/test_spidermiddleware_offsite.py @@ -73,7 +73,7 @@ class TestOffsiteMiddleware4(TestOffsiteMiddleware3): class TestOffsiteMiddleware5(TestOffsiteMiddleware4): - + def test_get_host_regex(self): self.spider.allowed_domains = ['http://scrapytest.org', 'scrapy.org', 'scrapy.test.org'] with warnings.catch_warnings(record=True) as w: diff --git a/tests/test_spidermiddleware_output_chain.py b/tests/test_spidermiddleware_output_chain.py index 5b7b5e7aa..739cf1c2d 100644 --- a/tests/test_spidermiddleware_output_chain.py +++ b/tests/test_spidermiddleware_output_chain.py @@ -156,7 +156,7 @@ class GeneratorFailMiddleware: r['processed'].append('{}.process_spider_output'.format(self.__class__.__name__)) yield r raise LookupError() - + def process_spider_exception(self, response, exception, spider): method = '{}.process_spider_exception'.format(self.__class__.__name__) spider.logger.info('%s: %s caught', method, exception.__class__.__name__) @@ -264,7 +264,7 @@ class TestSpiderMiddleware(TestCase): @classmethod def tearDownClass(cls): cls.mockserver.__exit__(None, None, None) - + @defer.inlineCallbacks def crawl_log(self, spider): crawler = get_crawler(spider) @@ -308,7 +308,7 @@ class TestSpiderMiddleware(TestCase): self.assertIn("{'from': 'errback'}", str(log1)) self.assertNotIn("{'from': 'callback'}", str(log1)) self.assertIn("'item_scraped_count': 1", str(log1)) - + @defer.inlineCallbacks def test_generator_callback(self): """ @@ -319,7 +319,7 @@ class TestSpiderMiddleware(TestCase): log2 = yield self.crawl_log(GeneratorCallbackSpider) self.assertIn("Middleware: ImportError exception caught", str(log2)) self.assertIn("'item_scraped_count': 2", str(log2)) - + @defer.inlineCallbacks def test_not_a_generator_callback(self): """ From c0d84f0962a4c269441db562e8cbc10298c53b72 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Wed, 18 Dec 2019 19:39:21 +0100 Subject: [PATCH 088/313] fix typos --- docs/contributing.rst | 2 +- docs/faq.rst | 2 +- docs/news.rst | 69 +++++++++++++++++----------------- docs/topics/jobs.rst | 2 +- docs/topics/leaks.rst | 2 +- docs/topics/media-pipeline.rst | 2 +- docs/topics/settings.rst | 2 +- docs/topics/telnetconsole.rst | 2 +- sep/sep-001.rst | 4 +- sep/sep-019.rst | 2 +- 10 files changed, 44 insertions(+), 45 deletions(-) diff --git a/docs/contributing.rst b/docs/contributing.rst index 3aebb3d50..b56295027 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -217,7 +217,7 @@ the tests with Python 3.6 use:: tox -e py36 -You can also specify a comma-separated list of environmets, and use :ref:`tox’s +You can also specify a comma-separated list of environments, and use :ref:`tox’s parallel mode <tox:parallel_mode>` to run the tests on multiple environments in parallel:: diff --git a/docs/faq.rst b/docs/faq.rst index 080d81981..aae2411e0 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -338,7 +338,7 @@ How to split an item into multiple items in an item pipeline? input item. :ref:`Create a spider middleware <custom-spider-middleware>` instead, and use its :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_spider_output` -method for this puspose. For example:: +method for this purpose. For example:: from copy import deepcopy diff --git a/docs/news.rst b/docs/news.rst index 9dfd28508..28db40e3c 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -678,7 +678,7 @@ Usability improvements * a message is added to IgnoreRequest in RobotsTxtMiddleware (:issue:`3113`) * better validation of ``url`` argument in ``Response.follow`` (:issue:`3131`) * non-zero exit code is returned from Scrapy commands when error happens - on spider inititalization (:issue:`3226`) + on spider initialization (:issue:`3226`) * Link extraction improvements: "ftp" is added to scheme list (:issue:`3152`); "flv" is added to common video extensions (:issue:`3165`) * better error message when an exporter is disabled (:issue:`3358`); @@ -1156,7 +1156,7 @@ Bug fixes - Fix :command:`view` command ; it was a regression in v1.3.0 (:issue:`2503`). - Fix tests regarding ``*_EXPIRES settings`` with Files/Images pipelines (:issue:`2460`). - Fix name of generated pipeline class when using basic project template (:issue:`2466`). -- Fix compatiblity with Twisted 17+ (:issue:`2496`, :issue:`2528`). +- Fix compatibility with Twisted 17+ (:issue:`2496`, :issue:`2528`). - Fix ``scrapy.Item`` inheritance on Python 3.6 (:issue:`2511`). - Enforce numeric values for components order in ``SPIDER_MIDDLEWARES``, ``DOWNLOADER_MIDDLEWARES``, ``EXTENIONS`` and ``SPIDER_CONTRACTS`` (:issue:`2420`). @@ -1164,7 +1164,7 @@ Bug fixes Documentation ~~~~~~~~~~~~~ -- Reword Code of Coduct section and upgrade to Contributor Covenant v1.4 +- Reword Code of Conduct section and upgrade to Contributor Covenant v1.4 (:issue:`2469`). - Clarify that passing spider arguments converts them to spider attributes (:issue:`2483`). @@ -1178,7 +1178,7 @@ Documentation Cleanups ~~~~~~~~ -- Remove reduntant check in ``MetaRefreshMiddleware`` (:issue:`2542`). +- Remove redundant check in ``MetaRefreshMiddleware`` (:issue:`2542`). - Faster checks in ``LinkExtractor`` for allow/deny patterns (:issue:`2538`). - Remove dead code supporting old Twisted versions (:issue:`2544`). @@ -1204,7 +1204,7 @@ New Features - ``MailSender`` now accepts single strings as values for ``to`` and ``cc`` arguments (:issue:`2272`) - ``scrapy fetch url``, ``scrapy shell url`` and ``fetch(url)`` inside - scrapy shell now follow HTTP redirections by default (:issue:`2290`); + Scrapy shell now follow HTTP redirections by default (:issue:`2290`); See :command:`fetch` and :command:`shell` for details. - ``HttpErrorMiddleware`` now logs errors with ``INFO`` level instead of ``DEBUG``; this is technically **backward incompatible** so please check your log parsers. @@ -1705,7 +1705,7 @@ Scrapy 1.0.4 (2015-12-30) - fix ValueError: Invalid XPath: //div/[id="not-exists"]/text() on selectors.rst (:commit:`ca8d60f`) - Typos corrections (:commit:`7067117`) - fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`) -- Add note to ubuntu install section about debian compatibility (:commit:`23fda69`) +- Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`) - Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`) - Reference Homebrew's homepage for installation instructions (:commit:`1925db1`) - Add oldest supported tox version to contributing docs (:commit:`5d10d6d`) @@ -1758,7 +1758,7 @@ Scrapy 1.0.1 (2015-07-01) - include tests/ to source distribution in MANIFEST.in (:commit:`eca227e`) - DOC Fix SelectJmes documentation (:commit:`b8567bc`) - DOC Bring Ubuntu and Archlinux outside of Windows subsection (:commit:`392233f`) -- DOC remove version suffix from ubuntu package (:commit:`5303c66`) +- DOC remove version suffix from Ubuntu package (:commit:`5303c66`) - DOC Update release date for 1.0 (:commit:`c89fa29`) .. _release-1.0.0: @@ -2211,7 +2211,7 @@ Scrapy 0.24.2 (2014-07-08) - Use a mutable mapping to proxy deprecated settings.overrides and settings.defaults attribute (:commit:`e5e8133`) - there is not support for python3 yet (:commit:`3cd6146`) -- Update python compatible version set to debian packages (:commit:`fa5d76b`) +- Update python compatible version set to Debian packages (:commit:`fa5d76b`) - DOC fix formatting in release notes (:commit:`c6a9e20`) Scrapy 0.24.1 (2014-06-27) @@ -2229,12 +2229,12 @@ Enhancements - Improve Scrapy top-level namespace (:issue:`494`, :issue:`684`) - Add selector shortcuts to responses (:issue:`554`, :issue:`690`) -- Add new lxml based LinkExtractor to replace unmantained SgmlLinkExtractor +- Add new lxml based LinkExtractor to replace unmaintained SgmlLinkExtractor (:issue:`559`, :issue:`761`, :issue:`763`) - Cleanup settings API - part of per-spider settings **GSoC project** (:issue:`737`) - Add UTF8 encoding header to templates (:issue:`688`, :issue:`762`) - Telnet console now binds to 127.0.0.1 by default (:issue:`699`) -- Update debian/ubuntu install instructions (:issue:`509`, :issue:`549`) +- Update Debian/Ubuntu install instructions (:issue:`509`, :issue:`549`) - Disable smart strings in lxml XPath evaluations (:issue:`535`) - Restore filesystem based cache as default for http cache middleware (:issue:`541`, :issue:`500`, :issue:`571`) @@ -2267,7 +2267,7 @@ Enhancements - Tests and docs for ``request_fingerprint`` function (:issue:`597`) - Update SEP-19 for GSoC project ``per-spider settings`` (:issue:`705`) - Set exit code to non-zero when contracts fails (:issue:`727`) -- Add a setting to control what class is instanciated as Downloader component +- Add a setting to control what class is instantiated as Downloader component (:issue:`738`) - Pass response in ``item_dropped`` signal (:issue:`724`) - Improve ``scrapy check`` contracts command (:issue:`733`, :issue:`752`) @@ -2276,7 +2276,7 @@ Enhancements - Add a note about reporting security issues (:issue:`697`) - Add LevelDB http cache storage backend (:issue:`626`, :issue:`500`) - Sort spider list output of ``scrapy list`` command (:issue:`742`) -- Multiple documentation enhancemens and fixes +- Multiple documentation enhancements and fixes (:issue:`575`, :issue:`587`, :issue:`590`, :issue:`596`, :issue:`610`, :issue:`617`, :issue:`618`, :issue:`627`, :issue:`613`, :issue:`643`, :issue:`654`, :issue:`675`, :issue:`663`, :issue:`711`, :issue:`714`) @@ -2321,19 +2321,19 @@ Scrapy 0.22.1 (released 2014-02-08) - BaseSgmlLinkExtractor: Added unit test of a link with an inner tag (:commit:`c1cb418`) - BaseSgmlLinkExtractor: Fixed unknown_endtag() so that it only set current_link=None when the end tag match the opening tag (:commit:`7e4d627`) - Fix tests for Travis-CI build (:commit:`76c7e20`) -- replace unencodeable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`) +- replace unencodable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`) - RegexLinkExtractor: encode URL unicode value when creating Links (:commit:`d0ee545`) - Updated the tutorial crawl output with latest output. (:commit:`8da65de`) - Updated shell docs with the crawler reference and fixed the actual shell output. (:commit:`875b9ab`) - PEP8 minor edits. (:commit:`f89efaf`) -- Expose current crawler in the scrapy shell. (:commit:`5349cec`) +- Expose current crawler in the Scrapy shell. (:commit:`5349cec`) - Unused re import and PEP8 minor edits. (:commit:`387f414`) - Ignore None's values when using the ItemLoader. (:commit:`0632546`) - DOC Fixed HTTPCACHE_STORAGE typo in the default value which is now Filesystem instead Dbm. (:commit:`cde9a8c`) -- show ubuntu setup instructions as literal code (:commit:`fb5c9c5`) +- show Ubuntu setup instructions as literal code (:commit:`fb5c9c5`) - Update Ubuntu installation instructions (:commit:`70fb105`) - Merge pull request #550 from stray-leone/patch-1 (:commit:`6f70b6a`) -- modify the version of scrapy ubuntu package (:commit:`725900d`) +- modify the version of Scrapy Ubuntu package (:commit:`725900d`) - fix 0.22.0 release date (:commit:`af0219a`) - fix typos in news.rst and remove (not released yet) header (:commit:`b7f58f4`) @@ -2354,7 +2354,7 @@ Enhancements - Improve test coverage and forthcoming Python 3 support (:issue:`525`) - Promote startup info on settings and middleware to INFO level (:issue:`520`) - Support partials in ``get_func_args`` util (:issue:`506`, issue:`504`) -- Allow running indiviual tests via tox (:issue:`503`) +- Allow running individual tests via tox (:issue:`503`) - Update extensions ignored by link extractors (:issue:`498`) - Add middleware methods to get files/images/thumbs paths (:issue:`490`) - Improve offsite middleware tests (:issue:`478`) @@ -2411,7 +2411,7 @@ Enhancements - scrapy.mail.MailSender now can connect over TLS or upgrade using STARTTLS (:issue:`327`) - New FilesPipeline with functionality factored out from ImagesPipeline (:issue:`370`, :issue:`409`) - Recommend Pillow instead of PIL for image handling (:issue:`317`) -- Added debian packages for Ubuntu quantal and raring (:commit:`86230c0`) +- Added Debian packages for Ubuntu Quantal and raring (:commit:`86230c0`) - Mock server (used for tests) can listen for HTTPS requests (:issue:`410`) - Remove multi spider support from multiple core components (:issue:`422`, :issue:`421`, :issue:`420`, :issue:`419`, :issue:`423`, :issue:`418`) @@ -2430,7 +2430,7 @@ Bugfixes - Fix tests under Django 1.6 (:commit:`b6bed44c`) - Lot of bugfixes to retry middleware under disconnections using HTTP 1.1 download handler - Fix inconsistencies among Twisted releases (:issue:`406`) -- Fix scrapy shell bugs (:issue:`418`, :issue:`407`) +- Fix Scrapy shell bugs (:issue:`418`, :issue:`407`) - Fix invalid variable name in setup.py (:issue:`429`) - Fix tutorial references (:issue:`387`) - Improve request-response docs (:issue:`391`) @@ -2512,15 +2512,15 @@ Scrapy 0.18.1 (released 2013-08-27) - test PotentiaDataLoss errors on unbound responses (:commit:`b15470d`) - Treat responses without content-length or Transfer-Encoding as good responses (:commit:`c4bf324`) - do no include ResponseFailed if http11 handler is not enabled (:commit:`6cbe684`) -- New HTTP client wraps connection losts in ResponseFailed exception. fix #373 (:commit:`1a20bba`) +- New HTTP client wraps connection lost in ResponseFailed exception. fix #373 (:commit:`1a20bba`) - limit travis-ci build matrix (:commit:`3b01bb8`) - Merge pull request #375 from peterarenot/patch-1 (:commit:`fa766d7`) - Fixed so it refers to the correct folder (:commit:`3283809`) -- added quantal & raring to support ubuntu releases (:commit:`1411923`) +- added Quantal & raring to support Ubuntu releases (:commit:`1411923`) - fix retry middleware which didn't retry certain connection errors after the upgrade to http1 client, closes GH-373 (:commit:`bb35ed0`) - fix XmlItemExporter in Python 2.7.4 and 2.7.5 (:commit:`de3e451`) - minor updates to 0.18 release notes (:commit:`c45e5f1`) -- fix contributters list format (:commit:`0b60031`) +- fix contributors list format (:commit:`0b60031`) Scrapy 0.18.0 (released 2013-08-09) ----------------------------------- @@ -2617,7 +2617,7 @@ contributors sorted by number of commits:: Scrapy 0.16.5 (released 2013-05-30) ----------------------------------- -- obey request method when scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`) +- obey request method when Scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`) - fix inaccurate downloader middleware documentation. refs #280 (:commit:`40667cb`) - doc: remove links to diveintopython.org, which is no longer available. closes #246 (:commit:`bd58bfa`) - Find form nodes in invalid html5 documents (:commit:`e3d6945`) @@ -2631,8 +2631,8 @@ Scrapy 0.16.4 (released 2013-01-23) - Fixed error message formatting. log.err() doesn't support cool formatting and when error occurred, the message was: "ERROR: Error processing %(item)s" (:commit:`c16150c`) - lint and improve images pipeline error logging (:commit:`56b45fc`) - fixed doc typos (:commit:`243be84`) -- add documentation topics: Broad Crawls & Common Practies (:commit:`1fbb715`) -- fix bug in scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`) +- add documentation topics: Broad Crawls & Common Practices (:commit:`1fbb715`) +- fix bug in Scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`) - Update docs/topics/commands.rst (:commit:`28eac7a`) Scrapy 0.16.3 (released 2012-12-07) @@ -2651,11 +2651,11 @@ Scrapy 0.16.3 (released 2012-12-07) Scrapy 0.16.2 (released 2012-11-09) ----------------------------------- -- scrapy contracts: python2.6 compat (:commit:`a4a9199`) -- scrapy contracts verbose option (:commit:`ec41673`) -- proper unittest-like output for scrapy contracts (:commit:`86635e4`) +- Scrapy contracts: python2.6 compat (:commit:`a4a9199`) +- Scrapy contracts verbose option (:commit:`ec41673`) +- proper unittest-like output for Scrapy contracts (:commit:`86635e4`) - added open_in_browser to debugging doc (:commit:`c9b690d`) -- removed reference to global scrapy stats from settings doc (:commit:`dd55067`) +- removed reference to global Scrapy stats from settings doc (:commit:`dd55067`) - Fix SpiderState bug in Windows platforms (:commit:`58998f4`) @@ -2665,7 +2665,7 @@ Scrapy 0.16.1 (released 2012-10-26) - fixed LogStats extension, which got broken after a wrong merge before the 0.16 release (:commit:`8c780fd`) - better backward compatibility for scrapy.conf.settings (:commit:`3403089`) - extended documentation on how to access crawler stats from extensions (:commit:`c4da0b5`) -- removed .hgtags (no longer needed now that scrapy uses git) (:commit:`d52c188`) +- removed .hgtags (no longer needed now that Scrapy uses git) (:commit:`d52c188`) - fix dashes under rst headers (:commit:`fa4f7f9`) - set release date for 0.16.0 in news (:commit:`e292246`) @@ -2680,8 +2680,7 @@ Scrapy changes: - documented :doc:`topics/autothrottle` and added to extensions installed by default. You still need to enable it with :setting:`AUTOTHROTTLE_ENABLED` - major Stats Collection refactoring: removed separation of global/per-spider stats, removed stats-related signals (``stats_spider_opened``, etc). Stats are much simpler now, backward compatibility is kept on the Stats Collector API and signals. - added :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_start_requests` method to spider middlewares -- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info. -- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info. +- dropped Signals singleton. Signals should now be accessed through the Crawler.signals attribute. See the signals documentation for more info. - dropped Stats Collector singleton. Stats can now be accessed through the Crawler.stats attribute. See the stats collection documentation for more info. - documented :ref:`topics-api` - ``lxml`` is now the default selectors backend instead of ``libxml2`` @@ -2715,7 +2714,7 @@ Scrapy changes: Scrapy 0.14.4 ------------- -- added precise to supported ubuntu distros (:commit:`b7e46df`) +- added precise to supported Ubuntu distros (:commit:`b7e46df`) - fixed bug in json-rpc webservice reported in https://groups.google.com/forum/#!topic/scrapy-users/qgVBmFybNAQ/discussion. also removed no longer supported 'run' command from extras/scrapy-ws.py (:commit:`340fbdb`) - meta tag attributes for content-type http equiv can be in any order. #123 (:commit:`0cb68af`) - replace "import Image" by more standard "from PIL import Image". closes #88 (:commit:`4d17048`) @@ -2728,11 +2727,11 @@ Scrapy 0.14.3 - include egg files used by testsuite in source distribution. #118 (:commit:`c897793`) - update docstring in project template to avoid confusion with genspider command, which may be considered as an advanced feature. refs #107 (:commit:`2548dcc`) - added note to docs/topics/firebug.rst about google directory being shut down (:commit:`668e352`) -- dont discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`) +- don't discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`) - do not fail handling unicode xpaths in libxml2 backed selectors (:commit:`b830e95`) - fixed minor mistake in Request objects documentation (:commit:`bf3c9ee`) - fixed minor defect in link extractors documentation (:commit:`ba14f38`) -- removed some obsolete remaining code related to sqlite support in scrapy (:commit:`0665175`) +- removed some obsolete remaining code related to sqlite support in Scrapy (:commit:`0665175`) Scrapy 0.14.2 ------------- diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst index f5542495b..8816a028c 100644 --- a/docs/topics/jobs.rst +++ b/docs/topics/jobs.rst @@ -74,7 +74,7 @@ Request serialization For persistence to work, :class:`~scrapy.http.Request` objects must be serializable with :mod:`pickle`, except for the ``callback`` and ``errback`` values passed to their ``__init__`` method, which must be methods of the -runnning :class:`~scrapy.spiders.Spider` class. +running :class:`~scrapy.spiders.Spider` class. If you wish to log the requests that couldn't be serialized, you can set the :setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page. diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 793636f59..83b4d8154 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -110,7 +110,7 @@ ties the response lifetime to the requests' one, and that would definitely cause memory leaks. Let's see how we can discover the cause (without knowing it -a-priori, of course) by using the ``trackref`` tool. +a priori, of course) by using the ``trackref`` tool. After the crawler is running for a few minutes and we notice its memory usage has grown a lot, we can enter its telnet console and check the live diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 206e7cfa5..a40682e5b 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -558,7 +558,7 @@ See here the methods that you can override in your custom Images Pipeline: Custom Images pipeline example ============================== -Here is a full example of the Images Pipeline whose methods are examplified +Here is a full example of the Images Pipeline whose methods are exemplified above:: import scrapy diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index a1d15a760..f138b7064 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -884,7 +884,7 @@ LOG_FORMAT Default: ``'%(asctime)s [%(name)s] %(levelname)s: %(message)s'`` -String for formatting log messsages. Refer to the `Python logging documentation`_ for the whole list of available +String for formatting log messages. Refer to the `Python logging documentation`_ for the whole list of available placeholders. .. _Python logging documentation: https://docs.python.org/2/library/logging.html#logrecord-attributes diff --git a/docs/topics/telnetconsole.rst b/docs/topics/telnetconsole.rst index 7db7e4f6b..ecdc5423c 100644 --- a/docs/topics/telnetconsole.rst +++ b/docs/topics/telnetconsole.rst @@ -48,7 +48,7 @@ autogenerated Password can be seen on scrapy logs like the example below:: 2018-10-16 14:35:21 [scrapy.extensions.telnet] INFO: Telnet Password: 16f92501e8a59326 -Default Username and Password can be overriden by the settings +Default Username and Password can be overridden by the settings :setting:`TELNETCONSOLE_USERNAME` and :setting:`TELNETCONSOLE_PASSWORD`. .. warning:: diff --git a/sep/sep-001.rst b/sep/sep-001.rst index 2a66f9802..00226283f 100644 --- a/sep/sep-001.rst +++ b/sep/sep-001.rst @@ -254,8 +254,8 @@ ItemForm #!python class MySiteForm(ItemForm): - witdth = adaptor(ItemForm.witdh, default_unit='cm') - volume = adaptor(ItemForm.witdh, default_unit='lt') + width = adaptor(ItemForm.width, default_unit='cm') + volume = adaptor(ItemForm.width, default_unit='lt') ia['width'] = x.x('//p[@class="width"]') ia['volume'] = x.x('//p[@class="volume"]') diff --git a/sep/sep-019.rst b/sep/sep-019.rst index 9fbf6a223..84f3a96c3 100644 --- a/sep/sep-019.rst +++ b/sep/sep-019.rst @@ -185,7 +185,7 @@ These ideas translate to the following changes on the ``SpiderManager`` class: will return a spider class, not an instance. It's basically a ``__get__`` to ``self._spiders``. -- All remaining functions should be deprecated or remove accordantly, since a +- All remaining functions should be deprecated or remove accordingly, since a crawler reference is no longer needed. - New helper ``get_spider_manager_class_from_scrapycfg`` in From 23a67cec271c849ad9b9c07a99531163dc3789fc Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Thu, 19 Dec 2019 09:57:17 +0100 Subject: [PATCH 089/313] fix first letter capitalization for Raring and Scrapy --- docs/contributing.rst | 2 +- docs/index.rst | 2 +- docs/intro/install.rst | 16 ++++++++-------- docs/news.rst | 20 ++++++++++---------- docs/topics/autothrottle.rst | 2 +- docs/topics/commands.rst | 2 +- docs/topics/contracts.rst | 2 +- docs/topics/debug.rst | 2 +- docs/topics/developer-tools.rst | 4 ++-- docs/topics/logging.rst | 4 ++-- docs/topics/settings.rst | 4 ++-- docs/topics/shell.rst | 2 +- docs/topics/telnetconsole.rst | 2 +- sep/sep-004.rst | 2 +- 14 files changed, 33 insertions(+), 33 deletions(-) diff --git a/docs/contributing.rst b/docs/contributing.rst index b56295027..f40a6bba2 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -44,7 +44,7 @@ guidelines when you're going to report a new bug. * check the :ref:`FAQ <faq>` first to see if your issue is addressed in a well-known question -* if you have a general question about scrapy usage, please ask it at +* if you have a general question about Scrapy usage, please ask it at `Stack Overflow <https://stackoverflow.com/questions/tagged/scrapy>`__ (use "scrapy" tag). diff --git a/docs/index.rst b/docs/index.rst index 6d5f9e77d..a4343b7e0 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -170,7 +170,7 @@ Solving specific problems Get answers to most frequently asked questions. :doc:`topics/debug` - Learn how to debug common problems of your scrapy spider. + Learn how to debug common problems of your Scrapy spider. :doc:`topics/contracts` Learn how to use contracts for testing your spiders. diff --git a/docs/intro/install.rst b/docs/intro/install.rst index e924b5303..0d6171884 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -78,9 +78,9 @@ TL;DR: We recommend installing Scrapy inside a virtual environment on all platforms. Python packages can be installed either globally (a.k.a system wide), -or in user-space. We do not recommend installing scrapy system wide. +or in user-space. We do not recommend installing Scrapy system wide. -Instead, we recommend that you install scrapy within a so-called +Instead, we recommend that you install Scrapy within a so-called "virtual environment" (`virtualenv`_). Virtualenvs allow you to not conflict with already-installed Python system packages (which could break some of your system tools and scripts), @@ -97,7 +97,7 @@ Check this `user guide`_ on how to create your virtualenv. .. note:: If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs. -Once you have created a virtualenv, you can install scrapy inside it with ``pip``, +Once you have created a virtualenv, you can install Scrapy inside it with ``pip``, just like any other Python package. (See :ref:`platform-specific guides <intro-install-platform-notes>` below for non-Python dependencies that you may need to install beforehand). @@ -144,7 +144,7 @@ albeit with potential issues with TLS connections. typically too old and slow to catch up with latest Scrapy. -To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install +To install Scrapy on Ubuntu (or Ubuntu-based) systems, you need to install these dependencies:: sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev @@ -225,17 +225,17 @@ PyPy We recommend using the latest PyPy version. The version tested is 5.9.0. For PyPy3, only Linux installation was tested. -Most scrapy dependencides now have binary wheels for CPython, but not for PyPy. +Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy. This means that these dependecies will be built during installation. On OS X, you are likely to face an issue with building Cryptography dependency, solution to this problem is described `here <https://github.com/pyca/cryptography/issues/2692#issuecomment-272773481>`_, that is to ``brew install openssl`` and then export the flags that this command -recommends (only needed when installing scrapy). Installing on Linux has no special +recommends (only needed when installing Scrapy). Installing on Linux has no special issues besides installing build dependencies. -Installing scrapy with PyPy on Windows is not tested. +Installing Scrapy with PyPy on Windows is not tested. -You can check that scrapy is installed correctly by running ``scrapy bench``. +You can check that Scrapy is installed correctly by running ``scrapy bench``. If this command gives errors such as ``TypeError: ... got 2 unexpected keyword arguments``, this means that setuptools was unable to pick up one PyPy-specific dependency. diff --git a/docs/news.rst b/docs/news.rst index 28db40e3c..406e889f5 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -1360,7 +1360,7 @@ Documentation - Grammar fixes: :issue:`2128`, :issue:`1566`. - Download stats badge removed from README (:issue:`2160`). -- New scrapy :ref:`architecture diagram <topics-architecture>` (:issue:`2165`). +- New Scrapy :ref:`architecture diagram <topics-architecture>` (:issue:`2165`). - Updated ``Response`` parameters documentation (:issue:`2197`). - Reworded misleading :setting:`RANDOMIZE_DOWNLOAD_DELAY` description (:issue:`2190`). - Add StackOverflow as a support channel (:issue:`2257`). @@ -1450,7 +1450,7 @@ Documentation - Use "url" variable in downloader middleware example (:issue:`2015`) - Grammar fixes (:issue:`2054`, :issue:`2120`) - New FAQ entry on using BeautifulSoup in spider callbacks (:issue:`2048`) -- Add notes about scrapy not working on Windows with Python 3 (:issue:`2060`) +- Add notes about Scrapy not working on Windows with Python 3 (:issue:`2060`) - Encourage complete titles in pull requests (:issue:`2026`) Tests @@ -1509,7 +1509,7 @@ This 1.1 release brings a lot of interesting features and bug fixes: You can use :setting:`FILES_STORE_S3_ACL` to change it. - We've reimplemented ``canonicalize_url()`` for more correct output, especially for URLs with non-ASCII characters (:issue:`1947`). - This could change link extractors output compared to previous scrapy versions. + This could change link extractors output compared to previous Scrapy versions. This may also invalidate some cache entries you could still have from pre-1.1 runs. **Warning: backward incompatible!**. @@ -1722,7 +1722,7 @@ Scrapy 1.0.4 (2015-12-30) - Merge pull request #1513 from mgedmin/patch-2 (:commit:`5d4daf8`) - Typo (:commit:`f8d0682`) - Fix list formatting (:commit:`5f83a93`) -- fix scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`) +- fix Scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`) - Merge pull request #1475 from rweindl/patch-1 (:commit:`2d688cd`) - Update tutorial.rst (:commit:`fbc1f25`) - Merge pull request #1449 from rhoekman/patch-1 (:commit:`7d6538c`) @@ -1734,7 +1734,7 @@ Scrapy 1.0.4 (2015-12-30) Scrapy 1.0.3 (2015-08-11) ------------------------- -- add service_identity to scrapy install_requires (:commit:`cbc2501`) +- add service_identity to Scrapy install_requires (:commit:`cbc2501`) - Workaround for travis#296 (:commit:`66af9cd`) .. _release-1.0.2: @@ -2411,7 +2411,7 @@ Enhancements - scrapy.mail.MailSender now can connect over TLS or upgrade using STARTTLS (:issue:`327`) - New FilesPipeline with functionality factored out from ImagesPipeline (:issue:`370`, :issue:`409`) - Recommend Pillow instead of PIL for image handling (:issue:`317`) -- Added Debian packages for Ubuntu Quantal and raring (:commit:`86230c0`) +- Added Debian packages for Ubuntu Quantal and Raring (:commit:`86230c0`) - Mock server (used for tests) can listen for HTTPS requests (:issue:`410`) - Remove multi spider support from multiple core components (:issue:`422`, :issue:`421`, :issue:`420`, :issue:`419`, :issue:`423`, :issue:`418`) @@ -2516,7 +2516,7 @@ Scrapy 0.18.1 (released 2013-08-27) - limit travis-ci build matrix (:commit:`3b01bb8`) - Merge pull request #375 from peterarenot/patch-1 (:commit:`fa766d7`) - Fixed so it refers to the correct folder (:commit:`3283809`) -- added Quantal & raring to support Ubuntu releases (:commit:`1411923`) +- added Quantal & Raring to support Ubuntu releases (:commit:`1411923`) - fix retry middleware which didn't retry certain connection errors after the upgrade to http1 client, closes GH-373 (:commit:`bb35ed0`) - fix XmlItemExporter in Python 2.7.4 and 2.7.5 (:commit:`de3e451`) - minor updates to 0.18 release notes (:commit:`c45e5f1`) @@ -2555,8 +2555,8 @@ Scrapy 0.18.0 (released 2013-08-09) - Collect idle downloader slots (:issue:`297`) - Add ``ftp://`` scheme downloader handler (:issue:`329`) - Added downloader benchmark webserver and spider tools :ref:`benchmarking` -- Moved persistent (on disk) queues to a separate project (queuelib_) which scrapy now depends on -- Add scrapy commands using external libraries (:issue:`260`) +- Moved persistent (on disk) queues to a separate project (queuelib_) which Scrapy now depends on +- Add Scrapy commands using external libraries (:issue:`260`) - Added ``--pdb`` option to ``scrapy`` command line tool - Added :meth:`XPathSelector.remove_namespaces <scrapy.selector.Selector.remove_namespaces>` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`. - Several improvements to spider contracts @@ -2568,7 +2568,7 @@ Scrapy 0.18.0 (released 2013-08-09) - several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez) - support custom download slots - added --spider option to "shell" command. -- log overridden settings when scrapy starts +- log overridden settings when Scrapy starts Thanks to everyone who contribute to this release. Here is a list of contributors sorted by number of commits:: diff --git a/docs/topics/autothrottle.rst b/docs/topics/autothrottle.rst index c9bece753..4317019fc 100644 --- a/docs/topics/autothrottle.rst +++ b/docs/topics/autothrottle.rst @@ -11,7 +11,7 @@ Design goals ============ 1. be nicer to sites instead of using default download delay of zero -2. automatically adjust scrapy to the optimum crawling speed, so the user +2. automatically adjust Scrapy to the optimum crawling speed, so the user doesn't have to tune the download delays to find the optimum one. The user only needs to specify the maximum concurrent requests it allows, and the extension does the rest. diff --git a/docs/topics/commands.rst b/docs/topics/commands.rst index 5b3cd7e75..a0dcba90d 100644 --- a/docs/topics/commands.rst +++ b/docs/topics/commands.rst @@ -29,7 +29,7 @@ in standard locations: 1. ``/etc/scrapy.cfg`` or ``c:\scrapy\scrapy.cfg`` (system-wide), 2. ``~/.config/scrapy.cfg`` (``$XDG_CONFIG_HOME``) and ``~/.scrapy.cfg`` (``$HOME``) for global (user-wide) settings, and -3. ``scrapy.cfg`` inside a scrapy project's root (see next section). +3. ``scrapy.cfg`` inside a Scrapy project's root (see next section). Settings from these files are merged in the listed order of preference: user-defined values have higher priority than system-wide defaults diff --git a/docs/topics/contracts.rst b/docs/topics/contracts.rst index 371ae62d5..43db8f101 100644 --- a/docs/topics/contracts.rst +++ b/docs/topics/contracts.rst @@ -64,7 +64,7 @@ Use the :command:`check` command to run the contract checks. Custom Contracts ================ -If you find you need more power than the built-in scrapy contracts you can +If you find you need more power than the built-in Scrapy contracts you can create and load your own contracts in the project by using the :setting:`SPIDER_CONTRACTS` setting:: diff --git a/docs/topics/debug.rst b/docs/topics/debug.rst index 4b2588518..d75f17301 100644 --- a/docs/topics/debug.rst +++ b/docs/topics/debug.rst @@ -5,7 +5,7 @@ Debugging Spiders ================= This document explains the most common techniques for debugging spiders. -Consider the following scrapy spider below:: +Consider the following Scrapy spider below:: import scrapy from myproject.items import MyItem diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index bf14643be..d1d4ebf5d 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -80,7 +80,7 @@ expand and collapse a tag by clicking on the arrow in front of it or by double clicking directly on the tag. If we expand the ``span`` tag with the ``class= "text"`` we will see the quote-text we clicked on. The `Inspector` lets you copy XPaths to selected elements. Let's try it out: Right-click on the ``span`` -tag, select ``Copy > XPath`` and paste it in the scrapy shell like so:: +tag, select ``Copy > XPath`` and paste it in the Scrapy shell like so:: $ scrapy shell "http://quotes.toscrape.com/" (...) @@ -159,7 +159,7 @@ The page is quite similar to the basic `quotes.toscrape.com`_-page, but instead of the above-mentioned ``Next`` button, the page automatically loads new quotes when you scroll to the bottom. We could go ahead and try out different XPaths directly, but instead -we'll check another quite useful command from the scrapy shell:: +we'll check another quite useful command from the Scrapy shell:: $ scrapy shell "quotes.toscrape.com/scroll" (...) diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index dd09477b8..d4d22d889 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -171,9 +171,9 @@ listed in `logging's logrecord attributes docs <https://docs.python.org/2/library/datetime.html#strftime-and-strptime-behavior>`_ respectively. -If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the scrapy +If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the Scrapy component that prints the log. It is unset by default, hence logs contain the -scrapy component responsible for that log output. +Scrapy component responsible for that log output. Command-line options -------------------- diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index f138b7064..afe4fade1 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1264,7 +1264,7 @@ Default:: 'scrapy.contracts.default.ScrapesContract': 3, } -A dict containing the scrapy contracts enabled by default in Scrapy. You should +A dict containing the Scrapy contracts enabled by default in Scrapy. You should never modify this setting in your project, modify :setting:`SPIDER_CONTRACTS` instead. For more info see :ref:`topics-contracts`. @@ -1295,7 +1295,7 @@ SPIDER_LOADER_WARN_ONLY Default: ``False`` -By default, when scrapy tries to import spider classes from :setting:`SPIDER_MODULES`, +By default, when Scrapy tries to import spider classes from :setting:`SPIDER_MODULES`, it will fail loudly if there is any ``ImportError`` exception. But you can choose to silence this exception and turn it into a simple warning by setting ``SPIDER_LOADER_WARN_ONLY = True``. diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index 68a0b19b5..4fe0dea06 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -31,7 +31,7 @@ for more info. Scrapy also has support for `bpython`_, and will try to use it where `IPython`_ is unavailable. -Through scrapy's settings you can configure it to use any one of +Through Scrapy's settings you can configure it to use any one of ``ipython``, ``bpython`` or the standard ``python`` shell, regardless of which are installed. This is done by setting the ``SCRAPY_PYTHON_SHELL`` environment variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`:: diff --git a/docs/topics/telnetconsole.rst b/docs/topics/telnetconsole.rst index ecdc5423c..47d8d393c 100644 --- a/docs/topics/telnetconsole.rst +++ b/docs/topics/telnetconsole.rst @@ -44,7 +44,7 @@ the console you need to type:: >>> By default Username is ``scrapy`` and Password is autogenerated. The -autogenerated Password can be seen on scrapy logs like the example below:: +autogenerated Password can be seen on Scrapy logs like the example below:: 2018-10-16 14:35:21 [scrapy.extensions.telnet] INFO: Telnet Password: 16f92501e8a59326 diff --git a/sep/sep-004.rst b/sep/sep-004.rst index 69edfa136..05b0eb99c 100644 --- a/sep/sep-004.rst +++ b/sep/sep-004.rst @@ -53,7 +53,7 @@ Here's a simple proof-of-concept code of such script: # ... do something more interesting with scraped_items ... The behaviour of the Scrapy crawler would be controller by the Scrapy settings, -naturally, just like any typical scrapy project. But the default settings +naturally, just like any typical Scrapy project. But the default settings should be sufficient so as to not require adding any specific setting. But, at the same time, you could do it if you need to, say, for specifying a custom middleware. From f6bc1940a3394f74d8aa088faf2912a43690b260 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 19 Dec 2019 12:06:15 +0100 Subject: [PATCH 090/313] Use Python 3.7 to build the documentation --- .readthedocs.yml | 4 +++- .travis.yml | 2 +- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/.readthedocs.yml b/.readthedocs.yml index 3c1c3e8be..563add75f 100644 --- a/.readthedocs.yml +++ b/.readthedocs.yml @@ -2,6 +2,8 @@ version: 2 sphinx: configuration: docs/conf.py python: - version: 3.8 + # For available versions, see: + # https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image + version: 3.7 # Keep in sync with .travis.yml install: - requirements: docs/requirements.txt diff --git a/.travis.yml b/.travis.yml index c870934e1..f9d0dc8be 100644 --- a/.travis.yml +++ b/.travis.yml @@ -25,7 +25,7 @@ matrix: - env: TOXENV=extra-deps python: 3.8 - env: TOXENV=docs - python: 3.8 + python: 3.7 # Keep in sync with .readthedocs.yml install: - | if [ "$TOXENV" = "pypy3" ]; then From e22c0c27d9d33383f4ac18a4142ccffe1d9a0d3b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 19 Dec 2019 12:15:54 +0100 Subject: [PATCH 091/313] Revert "Improve FilteringLinkExtractor.__new__" This reverts commit ee9881d2704798c9cd61b6da503bb0694227c58c. --- scrapy/linkextractors/__init__.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index d0d34035d..bdeab3a75 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -60,7 +60,7 @@ class FilteringLinkExtractor(object): warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' 'please use scrapy.linkextractors.LinkExtractor instead', ScrapyDeprecationWarning, stacklevel=2) - return super().__new__(cls, *args, **kwargs) + return super(FilteringLinkExtractor, cls).__new__(cls) def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains, restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text): From 40697dcbfa17dccf81adce7d033bc466ba6e98a2 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 20 Dec 2019 19:33:44 +0500 Subject: [PATCH 092/313] Remove deferred_from_coro from this PR. --- scrapy/utils/defer.py | 28 ---------------------------- 1 file changed, 28 deletions(-) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 530bf0e9d..20ce59297 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -1,14 +1,10 @@ """ Helper functions for dealing with Twisted deferreds """ -import asyncio -import inspect - from twisted.internet import defer, task from twisted.python import failure from scrapy.exceptions import IgnoreRequest -from scrapy.utils.asyncio import is_asyncio_reactor_installed def defer_fail(_failure): @@ -118,27 +114,3 @@ def iter_errback(iterable, errback, *a, **kw): break except Exception: errback(failure.Failure(), *a, **kw) - - -def _isfuture(o): - # workaround for Python before 3.5.3 not having asyncio.isfuture - if hasattr(asyncio, 'isfuture'): - return asyncio.isfuture(o) - return isinstance(o, asyncio.Future) - - -def deferred_from_coro(o, asyncio_enabled=False): - """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" - if isinstance(o, defer.Deferred): - return o - if _isfuture(o) or inspect.isawaitable(o): - if not asyncio_enabled: - # wrapping the coroutine directly into a Deferred, this doesn't work correctly with coroutines - # that use asyncio, e.g. "await asyncio.sleep(1)" - return defer.ensureDeferred(o) - else: - # wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor - if not is_asyncio_reactor_installed(): - raise TypeError('Using coroutines requires installing AsyncioSelectorReactor') - return defer.Deferred.fromFuture(asyncio.ensure_future(o)) - return o From e342de5038e3757660f947fe1fadf35b54cd7113 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 20 Dec 2019 19:37:50 +0500 Subject: [PATCH 093/313] Remove a stray newline. --- tests/mockserver.py | 1 - 1 file changed, 1 deletion(-) diff --git a/tests/mockserver.py b/tests/mockserver.py index d4e0362fb..a45277db9 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -6,7 +6,6 @@ from subprocess import Popen, PIPE from urllib.parse import urlencode from OpenSSL import SSL - from twisted.web.server import Site, NOT_DONE_YET from twisted.web.resource import Resource from twisted.web.static import File From 8de80f59db19d739a056a7e58662f90544fece16 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Sat, 21 Dec 2019 13:08:29 +0500 Subject: [PATCH 094/313] Raise an exception if ASYNCIO_ENABLED but the reactor is wrong. --- scrapy/crawler.py | 6 +++++- scrapy/utils/log.py | 8 +------- tests/test_crawler.py | 9 +++++---- 3 files changed, 11 insertions(+), 12 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 706c8a59d..a9443f7ac 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -14,7 +14,7 @@ from scrapy.extension import ExtensionManager from scrapy.settings import overridden_settings, Settings from scrapy.signalmanager import SignalManager from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.asyncio import install_asyncio_reactor +from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_reactor_installed from scrapy.utils.ossignal import install_shutdown_handlers, signal_names from scrapy.utils.misc import load_object from scrapy.utils.log import ( @@ -259,6 +259,10 @@ class CrawlerProcess(CrawlerRunner): super(CrawlerProcess, self).__init__(settings) if self.settings.getbool('ASYNCIO_ENABLED'): install_asyncio_reactor() + if not is_asyncio_reactor_installed(): + raise Exception("ASYNCIO_ENABLED is on but the Twisted asyncio " + "reactor is not installed, this is not supported.") + install_shutdown_handlers(self._signal_shutdown) configure_logging(self.settings, install_root_handler) log_scrapy_info(self.settings) diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index 0fe3d1549..6179e1bd1 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -11,7 +11,6 @@ from twisted.python import log as twisted_log import scrapy from scrapy.settings import Settings from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.asyncio import is_asyncio_reactor_installed from scrapy.utils.versions import scrapy_components_versions @@ -150,12 +149,7 @@ def log_scrapy_info(settings): for name, version in scrapy_components_versions() if name != "Scrapy")}) if settings.getbool('ASYNCIO_ENABLED'): - if is_asyncio_reactor_installed(): - logger.debug("Asyncio support enabled") - else: - logger.error("ASYNCIO_ENABLED is on but the Twisted asyncio " - "reactor is not installed, this is not supported " - "and asyncio coroutines will not work.") + logger.debug("Asyncio support enabled") class StreamLogger(object): diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 0b2645280..a2865fcd1 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -264,13 +264,14 @@ class CrawlerRunnerHasSpider(unittest.TestCase): @defer.inlineCallbacks def test_crawler_process_asyncio_enabled_true(self): with LogCapture(level=logging.DEBUG) as log: - runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': True}) - yield runner.crawl(NoRequestsSpider) if self.reactor_pytest == 'asyncio': + runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': True}) + yield runner.crawl(NoRequestsSpider) self.assertIn("Asyncio support enabled", str(log)) else: - self.assertNotIn("Asyncio support enabled", str(log)) - self.assertIn("ASYNCIO_ENABLED is on but the Twisted asyncio reactor is not installed", str(log)) + msg = "ASYNCIO_ENABLED is on but the Twisted asyncio reactor is not installed" + with self.assertRaisesRegex(Exception, msg): + runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': True}) @defer.inlineCallbacks def test_crawler_process_asyncio_enabled_false(self): From 931b7e68d33e06b624b49a7abeababb4e3540474 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 09:50:28 -0300 Subject: [PATCH 095/313] Update FileDownloadHandler test --- scrapy/core/downloader/handlers/file.py | 2 +- tests/test_downloader_handlers.py | 26 +++++++++++++------------ 2 files changed, 15 insertions(+), 13 deletions(-) diff --git a/scrapy/core/downloader/handlers/file.py b/scrapy/core/downloader/handlers/file.py index d445ba2e1..0d94e3df0 100644 --- a/scrapy/core/downloader/handlers/file.py +++ b/scrapy/core/downloader/handlers/file.py @@ -4,7 +4,7 @@ from scrapy.responsetypes import responsetypes from scrapy.utils.decorators import defers -class FileDownloadHandler(object): +class FileDownloadHandler: lazy = False @defers diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 4505f2bf7..ce4685eed 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -1,21 +1,20 @@ +import contextlib import os import shutil import tempfile from unittest import mock -import contextlib from testfixtures import LogCapture -from twisted.trial import unittest +from twisted.cred import checkers, credentials, portal +from twisted.internet import defer, error, reactor from twisted.protocols.policies import WrappingFactory from twisted.python.filepath import FilePath -from twisted.internet import reactor, defer, error -from twisted.web import server, static, util, resource +from twisted.trial import unittest +from twisted.web import resource, server, static, util from twisted.web._newclient import ResponseFailed from twisted.web.http import _DataLoss -from twisted.web.test.test_webclient import ForeverTakingResource, \ - NoLengthResource, HostHeaderResource, \ - PayloadResource -from twisted.cred import portal, checkers, credentials +from twisted.web.test.test_webclient import (ForeverTakingResource, HostHeaderResource, + NoLengthResource, PayloadResource) from w3lib.url import path_to_file_uri from scrapy.core.downloader.handlers import DownloadHandlers @@ -26,13 +25,14 @@ from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler from scrapy.core.downloader.handlers.s3 import S3DownloadHandler -from scrapy.spiders import Spider +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.http import Headers, Request from scrapy.http.response.text import TextResponse from scrapy.responsetypes import responsetypes -from scrapy.utils.test import get_crawler, skip_if_no_boto +from scrapy.spiders import Spider +from scrapy.utils.misc import create_instance from scrapy.utils.python import to_bytes -from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning +from scrapy.utils.test import get_crawler, skip_if_no_boto from tests.mockserver import MockServer, ssl_context_factory, Echo from tests.spiders import SingleRequestSpider @@ -117,7 +117,9 @@ class FileTestCase(unittest.TestCase): self.tmpname = self.mktemp() with open(self.tmpname + '^', 'w') as f: f.write('0123456789') - self.download_request = FileDownloadHandler().download_request + crawler = get_crawler() + handler = create_instance(FileDownloadHandler, crawler.settings, crawler) + self.download_request = handler.download_request def tearDown(self): os.unlink(self.tmpname + '^') From 342bf3cd35df856f4f2eafbf2717e9244c9deaf8 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 09:52:55 -0300 Subject: [PATCH 096/313] Explicit keyword arguments --- scrapy/core/downloader/handlers/__init__.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/scrapy/core/downloader/handlers/__init__.py b/scrapy/core/downloader/handlers/__init__.py index e8c4454d2..94e0e59ef 100644 --- a/scrapy/core/downloader/handlers/__init__.py +++ b/scrapy/core/downloader/handlers/__init__.py @@ -50,9 +50,9 @@ class DownloadHandlers(object): if skip_lazy and getattr(dhcls, 'lazy', True): return None dh = create_instance( - dhcls, - self._crawler.settings, - self._crawler, + objcls=dhcls, + settings=self._crawler.settings, + crawler=self._crawler, ) except NotConfigured as ex: self._notconfigured[scheme] = str(ex) From 9e5d945ef27ff9efee54b2245e833fefc3df72d8 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 09:55:47 -0300 Subject: [PATCH 097/313] Use create_instance in downloader handler tests --- tests/test_downloader_handlers.py | 35 ++++++++++++++++++++++++------- 1 file changed, 27 insertions(+), 8 deletions(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index ce4685eed..b63e8405e 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -252,7 +252,12 @@ class HttpTestCase(unittest.TestCase): else: self.port = reactor.listenTCP(0, self.wrapper, interface=self.host) self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls.from_crawler(get_crawler()) + crawler = get_crawler() + self.download_handler = create_instance( + objcls=self.download_handler_cls, + settings=crawler.settings, + crawler=crawler + ) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -492,8 +497,11 @@ class Http11TestCase(HttpTestCase): return self.test_download_broken_content_allow_data_loss('broken-chunked') def test_download_broken_content_allow_data_loss_via_setting(self, url='broken'): - download_handler = self.download_handler_cls.from_crawler( - get_crawler(settings_dict={'DOWNLOAD_FAIL_ON_DATALOSS': False}) + crawler = get_crawler(settings_dict={'DOWNLOAD_FAIL_ON_DATALOSS': False}) + download_handler = create_instance( + objcls=self.download_handler_cls, + settings=crawler.settings, + crawler=crawler ) request = Request(self.getURL(url)) d = download_handler.download_request(request, Spider('foo')) @@ -512,8 +520,11 @@ class Https11TestCase(Http11TestCase): @defer.inlineCallbacks def test_tls_logging(self): - download_handler = self.download_handler_cls.from_crawler( - get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True}) + crawler = get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True}) + download_handler = create_instance( + objcls=self.download_handler_cls, + settings=crawler.settings, + crawler=crawler ) try: with LogCapture() as log_capture: @@ -581,8 +592,11 @@ class Https11CustomCiphers(unittest.TestCase): 0, self.wrapper, ssl_context_factory(self.keyfile, self.certfile, cipher_string='CAMELLIA256-SHA'), interface=self.host) self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls.from_crawler( - get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'}) + crawler = get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'}) + self.download_handler = create_instance( + objcls=self.download_handler_cls, + settings=crawler.settings, + crawler=crawler ) self.download_request = self.download_handler.download_request @@ -679,7 +693,12 @@ class HttpProxyTestCase(unittest.TestCase): wrapper = WrappingFactory(site) self.port = reactor.listenTCP(0, wrapper, interface='127.0.0.1') self.portno = self.port.getHost().port - self.download_handler = self.download_handler_cls.from_crawler(get_crawler()) + crawler = get_crawler() + self.download_handler = create_instance( + objcls=self.download_handler_cls, + settings=crawler.settings, + crawler=crawler + ) self.download_request = self.download_handler.download_request @defer.inlineCallbacks From fa21d8687a0f94e5d79cb28af11312ad58629954 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 10:00:25 -0300 Subject: [PATCH 098/313] Use create_instance in S3DownloadHandler tests --- tests/test_downloader_handlers.py | 24 ++++++++++++++++++------ 1 file changed, 18 insertions(+), 6 deletions(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index b63e8405e..ac0e94364 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -776,10 +776,13 @@ class S3AnonTestCase(unittest.TestCase): def setUp(self): skip_if_no_boto() - self.s3reqh = S3DownloadHandler.from_crawler( - crawler=get_crawler(), + crawler = get_crawler() + self.s3reqh = create_instance( + objcls=S3DownloadHandler, + settings=crawler.settings, + crawler=crawler, httpdownloadhandler=HttpDownloadHandlerMock, - #anon=True, # is implicit + # anon=True, # implicit ) self.download_request = self.s3reqh.download_request self.spider = Spider('foo') @@ -805,8 +808,11 @@ class S3TestCase(unittest.TestCase): def setUp(self): skip_if_no_boto() - s3reqh = S3DownloadHandler.from_crawler( - crawler=get_crawler(), + crawler = get_crawler() + s3reqh = create_instance( + objcls=S3DownloadHandler, + settings=crawler.settings, + crawler=crawler, aws_access_key_id=self.AWS_ACCESS_KEY_ID, aws_secret_access_key=self.AWS_SECRET_ACCESS_KEY, httpdownloadhandler=HttpDownloadHandlerMock, @@ -830,7 +836,13 @@ class S3TestCase(unittest.TestCase): def test_extra_kw(self): try: - S3DownloadHandler.from_crawler(get_crawler(), extra_kw=True) + crawler = get_crawler() + create_instance( + objcls=S3DownloadHandler, + settings=crawler.settings, + crawler=crawler, + extra_kw=True, + ) except Exception as e: self.assertIsInstance(e, (TypeError, NotConfigured)) else: From 7e6387de407297a36dd0e9b8e2058ae809cf3123 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 10:02:58 -0300 Subject: [PATCH 099/313] Use create_instance in FTPDownloadHandler/DataURIDownloadHandler tests --- tests/test_downloader_handlers.py | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index ac0e94364..14d58b651 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -984,7 +984,8 @@ class BaseFTPTestCase(unittest.TestCase): self.factory = FTPFactory(portal=p) self.port = reactor.listenTCP(0, self.factory, interface="127.0.0.1") self.portNum = self.port.getHost().port - self.download_handler = FTPDownloadHandler(get_crawler()) + crawler = get_crawler() + self.download_handler = create_instance(FTPDownloadHandler, crawler.settings, crawler) self.addCleanup(self.port.stopListening) def tearDown(self): @@ -1098,7 +1099,8 @@ class AnonymousFTPTestCase(BaseFTPTestCase): userAnonymous=self.username) self.port = reactor.listenTCP(0, self.factory, interface="127.0.0.1") self.portNum = self.port.getHost().port - self.download_handler = FTPDownloadHandler(get_crawler()) + crawler = get_crawler() + self.download_handler = create_instance(FTPDownloadHandler, crawler.settings, crawler) self.addCleanup(self.port.stopListening) def tearDown(self): @@ -1108,7 +1110,8 @@ class AnonymousFTPTestCase(BaseFTPTestCase): class DataURITestCase(unittest.TestCase): def setUp(self): - self.download_handler = DataURIDownloadHandler() + crawler = get_crawler() + self.download_handler = create_instance(DataURIDownloadHandler, crawler.settings, crawler) self.download_request = self.download_handler.download_request self.spider = Spider('foo') From 8a567e98bbb1c7f917462d0376061303baef5883 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 10:05:49 -0300 Subject: [PATCH 100/313] Remove unnecessary __init__ methods in downloader handler tests --- tests/test_downloader_handlers.py | 23 +++++------------------ 1 file changed, 5 insertions(+), 18 deletions(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 14d58b651..b66b8151e 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -38,29 +38,16 @@ from tests.mockserver import MockServer, ssl_context_factory, Echo from tests.spiders import SingleRequestSpider -class DummyDH(object): +class DummyDH: lazy = False - def __init__(self, crawler): - pass - @classmethod - def from_crawler(cls, crawler): - return cls(crawler) - - -class DummyLazyDH(object): +class DummyLazyDH: # Default is lazy for backward compatibility - - def __init__(self, crawler): - pass - - @classmethod - def from_crawler(cls, crawler): - return cls(crawler) + pass -class OffDH(object): +class OffDH: lazy = False def __init__(self, crawler): @@ -765,7 +752,7 @@ class Http11ProxyTestCase(HttpProxyTestCase): class HttpDownloadHandlerMock(object): - def __init__(self, settings): + def __init__(self, settings, crawler): pass def download_request(self, request, spider): From a6ec89251eca6977f898e8341634bbe7288fd966 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 10:40:16 -0300 Subject: [PATCH 101/313] Downloader handlers: crawler=None in __init__ --- scrapy/core/downloader/handlers/ftp.py | 12 ++++++------ scrapy/core/downloader/handlers/http10.py | 8 ++++---- scrapy/core/downloader/handlers/http11.py | 8 +++----- scrapy/core/downloader/handlers/s3.py | 15 +++++++++------ tests/test_downloader_handlers.py | 4 +--- 5 files changed, 23 insertions(+), 24 deletions(-) diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index fafecc1a8..2b22465e0 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -63,7 +63,7 @@ class ReceivedDataProtocol(Protocol): _CODE_RE = re.compile(r"\d+") -class FTPDownloadHandler(object): +class FTPDownloadHandler: lazy = False CODE_MAPPING = { @@ -71,14 +71,14 @@ class FTPDownloadHandler(object): "default": 503, } - def __init__(self, crawler): - self.default_user = crawler.settings['FTP_USER'] - self.default_password = crawler.settings['FTP_PASSWORD'] - self.passive_mode = crawler.settings['FTP_PASSIVE_MODE'] + def __init__(self, settings, crawler=None): + self.default_user = settings['FTP_USER'] + self.default_password = settings['FTP_PASSWORD'] + self.passive_mode = settings['FTP_PASSIVE_MODE'] @classmethod def from_crawler(cls, crawler): - return cls(crawler) + return cls(crawler.settings, crawler) def download_request(self, request, spider): parsed_url = urlparse_cached(request) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index ce0801bcc..51c0acd1b 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -6,18 +6,18 @@ from scrapy.utils.misc import load_object, create_instance from scrapy.utils.python import to_unicode -class HTTP10DownloadHandler(object): +class HTTP10DownloadHandler: lazy = False - def __init__(self, crawler): + def __init__(self, settings, crawler=None): self.HTTPClientFactory = load_object(crawler.settings['DOWNLOADER_HTTPCLIENTFACTORY']) self.ClientContextFactory = load_object(crawler.settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) + self._settings = settings self._crawler = crawler - self._settings = crawler.settings @classmethod def from_crawler(cls, crawler): - return cls(crawler) + return cls(crawler.settings, crawler) def download_request(self, request, spider): """Return a deferred for the HTTP download""" diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 691937e97..25dc287df 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -28,12 +28,10 @@ from scrapy.utils.python import to_bytes, to_unicode logger = logging.getLogger(__name__) -class HTTP11DownloadHandler(object): +class HTTP11DownloadHandler: lazy = False - def __init__(self, crawler): - settings = crawler.settings - + def __init__(self, settings, crawler=None): self._pool = HTTPConnectionPool(reactor, persistent=True) self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN') self._pool._factory.noisy = False @@ -68,7 +66,7 @@ class HTTP11DownloadHandler(object): @classmethod def from_crawler(cls, crawler): - return cls(crawler) + return cls(crawler.settings, crawler) def download_request(self, request, spider): """Return a deferred for the HTTP download""" diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index b35b59f3a..93cad0662 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -4,6 +4,7 @@ from scrapy.core.downloader.handlers.http import HTTPDownloadHandler from scrapy.exceptions import NotConfigured from scrapy.utils.boto import is_botocore from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.misc import create_instance def _get_boto_connection(): @@ -30,14 +31,15 @@ def _get_boto_connection(): return _S3Connection -class S3DownloadHandler(object): +class S3DownloadHandler: - def __init__(self, crawler, aws_access_key_id=None, aws_secret_access_key=None, + def __init__(self, settings, crawler=None, + aws_access_key_id=None, aws_secret_access_key=None, httpdownloadhandler=HTTPDownloadHandler, **kw): if not aws_access_key_id: - aws_access_key_id = crawler.settings['AWS_ACCESS_KEY_ID'] + aws_access_key_id = settings['AWS_ACCESS_KEY_ID'] if not aws_secret_access_key: - aws_secret_access_key = crawler.settings['AWS_SECRET_ACCESS_KEY'] + aws_secret_access_key = settings['AWS_SECRET_ACCESS_KEY'] # If no credentials could be found anywhere, # consider this an anonymous connection request by default; @@ -66,11 +68,12 @@ class S3DownloadHandler(object): except Exception as ex: raise NotConfigured(str(ex)) - self._download_http = httpdownloadhandler(crawler).download_request + _http_handler = create_instance(httpdownloadhandler, settings, crawler) + self._download_http = _http_handler.download_request @classmethod def from_crawler(cls, crawler, *args, **kwargs): - return cls(crawler, *args, **kwargs) + return cls(crawler.settings, crawler, *args, **kwargs) def download_request(self, request, spider): p = urlparse_cached(request) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index b66b8151e..218360709 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -751,9 +751,7 @@ class Http11ProxyTestCase(HttpProxyTestCase): self.assertIn(domain, timeout.osError) -class HttpDownloadHandlerMock(object): - def __init__(self, settings, crawler): - pass +class HttpDownloadHandlerMock: def download_request(self, request, spider): return request From e2e15d66510c3040234ef9222ff2a17961c50eff Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 10:48:19 -0300 Subject: [PATCH 102/313] Downloader handlers: sort imports --- scrapy/core/downloader/handlers/__init__.py | 6 +++--- scrapy/core/downloader/handlers/datauri.py | 2 +- scrapy/core/downloader/handlers/ftp.py | 4 ++-- scrapy/core/downloader/handlers/http10.py | 2 +- scrapy/core/downloader/handlers/http11.py | 20 ++++++++++---------- 5 files changed, 17 insertions(+), 17 deletions(-) diff --git a/scrapy/core/downloader/handlers/__init__.py b/scrapy/core/downloader/handlers/__init__.py index 94e0e59ef..e86680978 100644 --- a/scrapy/core/downloader/handlers/__init__.py +++ b/scrapy/core/downloader/handlers/__init__.py @@ -4,17 +4,17 @@ import logging from twisted.internet import defer -from scrapy.exceptions import NotSupported, NotConfigured +from scrapy import signals +from scrapy.exceptions import NotConfigured, NotSupported from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.misc import create_instance, load_object from scrapy.utils.python import without_none_values -from scrapy import signals logger = logging.getLogger(__name__) -class DownloadHandlers(object): +class DownloadHandlers: def __init__(self, crawler): self._crawler = crawler diff --git a/scrapy/core/downloader/handlers/datauri.py b/scrapy/core/downloader/handlers/datauri.py index 97134e618..a45b4ff3c 100644 --- a/scrapy/core/downloader/handlers/datauri.py +++ b/scrapy/core/downloader/handlers/datauri.py @@ -5,7 +5,7 @@ from scrapy.responsetypes import responsetypes from scrapy.utils.decorators import defers -class DataURIDownloadHandler(object): +class DataURIDownloadHandler: lazy = False @defers diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 2b22465e0..89c88ad70 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -33,8 +33,8 @@ from io import BytesIO from urllib.parse import unquote from twisted.internet import reactor -from twisted.protocols.ftp import FTPClient, CommandFailed -from twisted.internet.protocol import Protocol, ClientCreator +from twisted.internet.protocol import ClientCreator, Protocol +from twisted.protocols.ftp import CommandFailed, FTPClient from scrapy.http import Response from scrapy.responsetypes import responsetypes diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index 51c0acd1b..1086a6cc0 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -2,7 +2,7 @@ """ from twisted.internet import reactor -from scrapy.utils.misc import load_object, create_instance +from scrapy.utils.misc import create_instance, load_object from scrapy.utils.python import to_unicode diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 25dc287df..d8a561792 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -1,27 +1,27 @@ """Download handlers for http and https schemes""" -import re import logging +import re import warnings from io import BytesIO from time import time from urllib.parse import urldefrag -from zope.interface import implementer -from twisted.internet import defer, reactor, protocol +from twisted.internet import defer, protocol, reactor +from twisted.internet.endpoints import TCP4ClientEndpoint +from twisted.internet.error import TimeoutError +from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI +from twisted.web.http import _DataLoss, PotentialDataLoss from twisted.web.http_headers import Headers as TxHeaders from twisted.web.iweb import IBodyProducer, UNKNOWN_LENGTH -from twisted.internet.error import TimeoutError -from twisted.web.http import _DataLoss, PotentialDataLoss -from twisted.web.client import Agent, ResponseDone, HTTPConnectionPool, ResponseFailed, URI -from twisted.internet.endpoints import TCP4ClientEndpoint +from zope.interface import implementer +from scrapy.core.downloader.tls import openssl_methods +from scrapy.core.downloader.webclient import _parse from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Headers from scrapy.responsetypes import responsetypes -from scrapy.core.downloader.webclient import _parse -from scrapy.core.downloader.tls import openssl_methods -from scrapy.utils.misc import load_object, create_instance +from scrapy.utils.misc import create_instance, load_object from scrapy.utils.python import to_bytes, to_unicode From 2fb160e3bac9e09373a49cb0b2d764ddf782987c Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 20:24:16 -0300 Subject: [PATCH 103/313] Use settings instead of crawler --- scrapy/core/downloader/handlers/ftp.py | 4 ++-- scrapy/core/downloader/handlers/http10.py | 4 ++-- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 89c88ad70..1681c6df8 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -71,14 +71,14 @@ class FTPDownloadHandler: "default": 503, } - def __init__(self, settings, crawler=None): + def __init__(self, settings): self.default_user = settings['FTP_USER'] self.default_password = settings['FTP_PASSWORD'] self.passive_mode = settings['FTP_PASSIVE_MODE'] @classmethod def from_crawler(cls, crawler): - return cls(crawler.settings, crawler) + return cls(crawler.settings) def download_request(self, request, spider): parsed_url = urlparse_cached(request) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index 1086a6cc0..87a42f1da 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -10,8 +10,8 @@ class HTTP10DownloadHandler: lazy = False def __init__(self, settings, crawler=None): - self.HTTPClientFactory = load_object(crawler.settings['DOWNLOADER_HTTPCLIENTFACTORY']) - self.ClientContextFactory = load_object(crawler.settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) + self.HTTPClientFactory = load_object(settings['DOWNLOADER_HTTPCLIENTFACTORY']) + self.ClientContextFactory = load_object(settings['DOWNLOADER_CLIENTCONTEXTFACTORY']) self._settings = settings self._crawler = crawler From 9a75b46fb8322d27ffd45ee6c187bc84a565e26d Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 20:26:58 -0300 Subject: [PATCH 104/313] Explicit argument names --- scrapy/core/downloader/handlers/http10.py | 2 +- scrapy/core/downloader/handlers/http11.py | 4 ++-- scrapy/core/downloader/handlers/s3.py | 6 +++++- 3 files changed, 8 insertions(+), 4 deletions(-) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index 87a42f1da..d4aa51bd1 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -29,7 +29,7 @@ class HTTP10DownloadHandler: host, port = to_unicode(factory.host), factory.port if factory.scheme == b'https': client_context_factory = create_instance( - self.ClientContextFactory, + objcls=self.ClientContextFactory, settings=self._settings, crawler=self._crawler, ) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index d8a561792..5a5f6cf0a 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -41,7 +41,7 @@ class HTTP11DownloadHandler: # try method-aware context factory try: self._contextFactory = create_instance( - self._contextFactoryClass, + objcls=self._contextFactoryClass, settings=settings, crawler=crawler, method=self._sslMethod, @@ -49,7 +49,7 @@ class HTTP11DownloadHandler: except TypeError: # use context factory defaults self._contextFactory = create_instance( - self._contextFactoryClass, + objcls=self._contextFactoryClass, settings=settings, crawler=crawler, ) diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index 93cad0662..b38d2bf86 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -68,7 +68,11 @@ class S3DownloadHandler: except Exception as ex: raise NotConfigured(str(ex)) - _http_handler = create_instance(httpdownloadhandler, settings, crawler) + _http_handler = create_instance( + objcls=httpdownloadhandler, + settings=settings, + crawler=crawler, + ) self._download_http = _http_handler.download_request @classmethod From 982a66f9fb627575d42f0ad4fb2eb38b0a55b784 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 20:28:17 -0300 Subject: [PATCH 105/313] [test] Download handler: avoid passing settings if not necessary --- tests/test_downloader_handlers.py | 41 +++++++------------------------ 1 file changed, 9 insertions(+), 32 deletions(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 218360709..8d95d7cac 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -104,8 +104,7 @@ class FileTestCase(unittest.TestCase): self.tmpname = self.mktemp() with open(self.tmpname + '^', 'w') as f: f.write('0123456789') - crawler = get_crawler() - handler = create_instance(FileDownloadHandler, crawler.settings, crawler) + handler = create_instance(FileDownloadHandler, None, get_crawler()) self.download_request = handler.download_request def tearDown(self): @@ -239,12 +238,7 @@ class HttpTestCase(unittest.TestCase): else: self.port = reactor.listenTCP(0, self.wrapper, interface=self.host) self.portno = self.port.getHost().port - crawler = get_crawler() - self.download_handler = create_instance( - objcls=self.download_handler_cls, - settings=crawler.settings, - crawler=crawler - ) + self.download_handler = create_instance(self.download_handler_cls, None, get_crawler()) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -485,11 +479,7 @@ class Http11TestCase(HttpTestCase): def test_download_broken_content_allow_data_loss_via_setting(self, url='broken'): crawler = get_crawler(settings_dict={'DOWNLOAD_FAIL_ON_DATALOSS': False}) - download_handler = create_instance( - objcls=self.download_handler_cls, - settings=crawler.settings, - crawler=crawler - ) + download_handler = create_instance(self.download_handler_cls, None, crawler) request = Request(self.getURL(url)) d = download_handler.download_request(request, Spider('foo')) d.addCallback(lambda r: r.flags) @@ -508,11 +498,7 @@ class Https11TestCase(Http11TestCase): @defer.inlineCallbacks def test_tls_logging(self): crawler = get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING': True}) - download_handler = create_instance( - objcls=self.download_handler_cls, - settings=crawler.settings, - crawler=crawler - ) + download_handler = create_instance(self.download_handler_cls, None, crawler) try: with LogCapture() as log_capture: request = Request(self.getURL('file')) @@ -580,11 +566,7 @@ class Https11CustomCiphers(unittest.TestCase): interface=self.host) self.portno = self.port.getHost().port crawler = get_crawler(settings_dict={'DOWNLOADER_CLIENT_TLS_CIPHERS': 'CAMELLIA256-SHA'}) - self.download_handler = create_instance( - objcls=self.download_handler_cls, - settings=crawler.settings, - crawler=crawler - ) + self.download_handler = create_instance(self.download_handler_cls, None, crawler) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -680,12 +662,7 @@ class HttpProxyTestCase(unittest.TestCase): wrapper = WrappingFactory(site) self.port = reactor.listenTCP(0, wrapper, interface='127.0.0.1') self.portno = self.port.getHost().port - crawler = get_crawler() - self.download_handler = create_instance( - objcls=self.download_handler_cls, - settings=crawler.settings, - crawler=crawler - ) + self.download_handler = create_instance(self.download_handler_cls, None, get_crawler()) self.download_request = self.download_handler.download_request @defer.inlineCallbacks @@ -764,7 +741,7 @@ class S3AnonTestCase(unittest.TestCase): crawler = get_crawler() self.s3reqh = create_instance( objcls=S3DownloadHandler, - settings=crawler.settings, + settings=None, crawler=crawler, httpdownloadhandler=HttpDownloadHandlerMock, # anon=True, # implicit @@ -796,7 +773,7 @@ class S3TestCase(unittest.TestCase): crawler = get_crawler() s3reqh = create_instance( objcls=S3DownloadHandler, - settings=crawler.settings, + settings=None, crawler=crawler, aws_access_key_id=self.AWS_ACCESS_KEY_ID, aws_secret_access_key=self.AWS_SECRET_ACCESS_KEY, @@ -824,7 +801,7 @@ class S3TestCase(unittest.TestCase): crawler = get_crawler() create_instance( objcls=S3DownloadHandler, - settings=crawler.settings, + settings=None, crawler=crawler, extra_kw=True, ) From ab54e0d33e2a461059a9f1e9759c95b18e38da28 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 23 Dec 2019 20:37:18 -0300 Subject: [PATCH 106/313] Keyword-only args for S3DownloadHandler --- scrapy/core/downloader/handlers/s3.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index b38d2bf86..40a1fa48e 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -33,7 +33,8 @@ def _get_boto_connection(): class S3DownloadHandler: - def __init__(self, settings, crawler=None, + def __init__(self, settings, *, + crawler=None, aws_access_key_id=None, aws_secret_access_key=None, httpdownloadhandler=HTTPDownloadHandler, **kw): if not aws_access_key_id: @@ -76,8 +77,8 @@ class S3DownloadHandler: self._download_http = _http_handler.download_request @classmethod - def from_crawler(cls, crawler, *args, **kwargs): - return cls(crawler.settings, crawler, *args, **kwargs) + def from_crawler(cls, crawler, **kwargs): + return cls(crawler.settings, crawler=crawler, **kwargs) def download_request(self, request, spider): p = urlparse_cached(request) From 87ece066ca320b07acda57c99aee8a62992ec144 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 26 Dec 2019 20:41:06 +0500 Subject: [PATCH 107/313] Remove conditional asyncio imports. --- scrapy/utils/asyncio.py | 16 ++++------------ 1 file changed, 4 insertions(+), 12 deletions(-) diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py index b53c8a8b0..917973de2 100644 --- a/scrapy/utils/asyncio.py +++ b/scrapy/utils/asyncio.py @@ -1,25 +1,17 @@ +import asyncio from contextlib import suppress +from twisted.internet import asyncioreactor from twisted.internet.error import ReactorAlreadyInstalledError def install_asyncio_reactor(): """ Tries to install AsyncioSelectorReactor """ - try: - import asyncio - from twisted.internet import asyncioreactor - except ImportError: - return - with suppress(ReactorAlreadyInstalledError): asyncioreactor.install(asyncio.get_event_loop()) def is_asyncio_reactor_installed(): - try: - import twisted.internet.reactor - from twisted.internet import asyncioreactor - return isinstance(twisted.internet.reactor, asyncioreactor.AsyncioSelectorReactor) - except ImportError: - return False + from twisted.internet import reactor + return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor) From 37ac47ff8074959ea66566fcb8b0e9e62272f963 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 26 Dec 2019 20:46:54 +0500 Subject: [PATCH 108/313] Fix a deprecation warning. --- tests/test_utils_asyncio.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_utils_asyncio.py b/tests/test_utils_asyncio.py index a6ba24876..44acc24af 100644 --- a/tests/test_utils_asyncio.py +++ b/tests/test_utils_asyncio.py @@ -10,7 +10,7 @@ class AsyncioTest(TestCase): def test_is_asyncio_reactor_installed(self): # the result should depend only on the pytest --reactor argument - self.assertEquals(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio') + self.assertEqual(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio') def test_install_asyncio_reactor(self): # this should do nothing From 30ebd05a5f627262702ad1e1e488d6176a8c7882 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 27 Dec 2019 00:05:14 +0500 Subject: [PATCH 109/313] Simplify the tox asyncio entries. --- tox.ini | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tox.ini b/tox.ini index ed0d4c9ab..b62100026 100644 --- a/tox.ini +++ b/tox.ini @@ -99,7 +99,7 @@ commands = [asyncio] commands = - py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests} + {[testenv]commands} --reactor=asyncio [testenv:py35-asyncio] basepython = python3.5 From f75ccc997aa75fadf96a8d4b836248397ef89802 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 27 Dec 2019 19:48:54 +0500 Subject: [PATCH 110/313] FIx a typo in the only_asyncio fixture. --- conftest.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/conftest.py b/conftest.py index 6d9696a3f..c0de09909 100644 --- a/conftest.py +++ b/conftest.py @@ -48,4 +48,4 @@ def reactor_pytest(request): @pytest.fixture(autouse=True) def only_asyncio(request, reactor_pytest): if request.node.get_closest_marker('only_asyncio') and reactor_pytest != 'asyncio': - pytest.skip('This test is only run with --reactor-asyncio') + pytest.skip('This test is only run with --reactor=asyncio') From dc1ee09481c7655a8ebf77a75ccc965a4ba5400d Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 27 Dec 2019 21:55:58 +0500 Subject: [PATCH 111/313] Rename ASYNCIO_ENABLED to ASYNCIO_REACTOR, change the logic accordingly. --- docs/topics/settings.rst | 14 +++---- scrapy/crawler.py | 17 +++++--- scrapy/settings/default_settings.py | 2 +- scrapy/utils/log.py | 5 ++- .../asyncio_enabled_no_reactor.py | 2 +- .../CrawlerProcess/asyncio_enabled_reactor.py | 2 +- tests/test_commands.py | 8 ++-- tests/test_crawler.py | 42 ++++++++----------- 8 files changed, 43 insertions(+), 49 deletions(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 62b2870b2..c02f877fc 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -160,26 +160,22 @@ to any particular component. In that case the module of that component will be shown, typically an extension, middleware or pipeline. It also means that the component must be enabled in order for the setting to have any effect. -.. setting:: ASYNCIO_ENABLED +.. setting:: ASYNCIO_REACTOR -ASYNCIO_ENABLED +ASYNCIO_REACTOR --------------- Default: ``False`` -Whether to support ``async def`` methods and callbacks which use code that -requires an asyncio loop. - -If an ``async def`` coroutine doesn't require the asyncio loop, it will work -even if this is set to ``False``. Coroutines that require the asyncio loop may -silently fail to run or raise errors unless this is set to ``True``. +Whether to install and require the Twisted reactor that uses the asyncio loop. When this option is set to ``True``, Scrapy will require :class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`. It will install this reactor if no reactor is installed yet, such as when using the ``scrapy`` script or :class:`~scrapy.crawler.CrawlerProcess`. If you are using :class:`~scrapy.crawler.CrawlerRunner`, you need to install the correct reactor -manually. +manually. If a different reactor is installed outside Scrapy, it will raise an +exception. The default value for this option is currently ``False`` to maintain backward compatibility and avoid possible problems caused by using a different Twisted diff --git a/scrapy/crawler.py b/scrapy/crawler.py index a9443f7ac..f87e67d93 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -137,6 +137,7 @@ class CrawlerRunner(object): self._crawlers = set() self._active = set() self.bootstrap_failed = False + self._handle_asyncio_reactor() @property def spiders(self): @@ -230,6 +231,11 @@ class CrawlerRunner(object): while self._active: yield defer.DeferredList(self._active) + def _handle_asyncio_reactor(self): + if self.settings.getbool('ASYNCIO_REACTOR') and not is_asyncio_reactor_installed(): + raise Exception("ASYNCIO_REACTOR is on but the Twisted asyncio " + "reactor is not installed.") + class CrawlerProcess(CrawlerRunner): """ @@ -257,12 +263,6 @@ class CrawlerProcess(CrawlerRunner): def __init__(self, settings=None, install_root_handler=True): super(CrawlerProcess, self).__init__(settings) - if self.settings.getbool('ASYNCIO_ENABLED'): - install_asyncio_reactor() - if not is_asyncio_reactor_installed(): - raise Exception("ASYNCIO_ENABLED is on but the Twisted asyncio " - "reactor is not installed, this is not supported.") - install_shutdown_handlers(self._signal_shutdown) configure_logging(self.settings, install_root_handler) log_scrapy_info(self.settings) @@ -333,6 +333,11 @@ class CrawlerProcess(CrawlerRunner): except RuntimeError: # raised if already stopped or in shutdown stage pass + def _handle_asyncio_reactor(self): + if self.settings.getbool('ASYNCIO_REACTOR'): + install_asyncio_reactor() + super()._handle_asyncio_reactor() + def _get_spider_loader(settings): """ Get SpiderLoader instance from settings """ diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index a7792e248..d03fd37b0 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -19,7 +19,7 @@ from os.path import join, abspath, dirname AJAXCRAWL_ENABLED = False -ASYNCIO_ENABLED = False +ASYNCIO_REACTOR = False AUTOTHROTTLE_ENABLED = False AUTOTHROTTLE_DEBUG = False diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index 6179e1bd1..e4cf0196b 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -11,6 +11,7 @@ from twisted.python import log as twisted_log import scrapy from scrapy.settings import Settings from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.asyncio import is_asyncio_reactor_installed from scrapy.utils.versions import scrapy_components_versions @@ -148,8 +149,8 @@ def log_scrapy_info(settings): {'versions': ", ".join("%s %s" % (name, version) for name, version in scrapy_components_versions() if name != "Scrapy")}) - if settings.getbool('ASYNCIO_ENABLED'): - logger.debug("Asyncio support enabled") + if is_asyncio_reactor_installed(): + logger.debug("Asyncio reactor is installed") class StreamLogger(object): diff --git a/tests/CrawlerProcess/asyncio_enabled_no_reactor.py b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py index dfe028ef4..db1b75931 100644 --- a/tests/CrawlerProcess/asyncio_enabled_no_reactor.py +++ b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py @@ -10,7 +10,7 @@ class NoRequestsSpider(scrapy.Spider): process = CrawlerProcess(settings={ - 'ASYNCIO_ENABLED': True, + 'ASYNCIO_REACTOR': True, }) process.crawl(NoRequestsSpider) diff --git a/tests/CrawlerProcess/asyncio_enabled_reactor.py b/tests/CrawlerProcess/asyncio_enabled_reactor.py index 7a172ea28..cec3c9c25 100644 --- a/tests/CrawlerProcess/asyncio_enabled_reactor.py +++ b/tests/CrawlerProcess/asyncio_enabled_reactor.py @@ -15,7 +15,7 @@ class NoRequestsSpider(scrapy.Spider): process = CrawlerProcess(settings={ - 'ASYNCIO_ENABLED': True, + 'ASYNCIO_REACTOR': True, }) process.crawl(NoRequestsSpider) diff --git a/tests/test_commands.py b/tests/test_commands.py index 197d80217..6024af71c 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -296,12 +296,12 @@ class BadSpider(scrapy.Spider): self.assertIn("badspider.py", log) def test_asyncio_enabled_true(self): - log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_ENABLED=True']) - self.assertIn("DEBUG: Asyncio support enabled", log) + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_REACTOR=True']) + self.assertIn("DEBUG: Asyncio reactor is installed", log) def test_asyncio_enabled_false(self): - log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_ENABLED=False']) - self.assertNotIn("DEBUG: Asyncio support enabled", log) + log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_REACTOR=False']) + self.assertNotIn("DEBUG: Asyncio reactor is installed", log) class BenchCommandTest(CommandTest): diff --git a/tests/test_crawler.py b/tests/test_crawler.py index a2865fcd1..fce60ca37 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -13,7 +13,6 @@ import scrapy from scrapy.crawler import Crawler, CrawlerRunner, CrawlerProcess from scrapy.settings import Settings, default_settings from scrapy.spiderloader import SpiderLoader -from scrapy.utils.asyncio import is_asyncio_reactor_installed from scrapy.utils.log import configure_logging, get_scrapy_root_handler from scrapy.utils.spider import DefaultSpider from scrapy.utils.misc import load_object @@ -209,14 +208,6 @@ class NoRequestsSpider(scrapy.Spider): return [] -class AsyncioSpider(scrapy.Spider): - name = 'asyncio' - - def start_requests(self): - self.logger.info('Asyncio support: %s', is_asyncio_reactor_installed()) - return [] - - @mark.usefixtures('reactor_pytest') class CrawlerRunnerHasSpider(unittest.TestCase): @@ -261,31 +252,32 @@ class CrawlerRunnerHasSpider(unittest.TestCase): self.assertEqual(runner.bootstrap_failed, True) + def test_crawler_runner_asyncio_enabled_true(self): + if self.reactor_pytest == 'asyncio': + runner = CrawlerRunner(settings={'ASYNCIO_REACTOR': True}) + else: + msg = "ASYNCIO_REACTOR is on but the Twisted asyncio reactor is not installed" + with self.assertRaisesRegex(Exception, msg): + runner = CrawlerRunner(settings={'ASYNCIO_REACTOR': True}) + @defer.inlineCallbacks def test_crawler_process_asyncio_enabled_true(self): with LogCapture(level=logging.DEBUG) as log: if self.reactor_pytest == 'asyncio': - runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': True}) + runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': True}) yield runner.crawl(NoRequestsSpider) - self.assertIn("Asyncio support enabled", str(log)) + self.assertIn("Asyncio reactor is installed", str(log)) else: - msg = "ASYNCIO_ENABLED is on but the Twisted asyncio reactor is not installed" + msg = "ASYNCIO_REACTOR is on but the Twisted asyncio reactor is not installed" with self.assertRaisesRegex(Exception, msg): - runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': True}) + runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': True}) @defer.inlineCallbacks def test_crawler_process_asyncio_enabled_false(self): - runner = CrawlerProcess(settings={'ASYNCIO_ENABLED': False}) + runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': False}) with LogCapture(level=logging.DEBUG) as log: yield runner.crawl(NoRequestsSpider) - self.assertNotIn("Asyncio support enabled", str(log)) - - @defer.inlineCallbacks - def test_crawler_runner_asyncio_supported(self): - runner = CrawlerRunner() - with LogCapture() as log: - yield runner.crawl(AsyncioSpider) - log.check_present(('asyncio', 'INFO', 'Asyncio support: %s' % (self.reactor_pytest == 'asyncio'))) + self.assertNotIn("Asyncio reactor is installed", str(log)) class CrawlerProcessSubprocess(unittest.TestCase): @@ -302,14 +294,14 @@ class CrawlerProcessSubprocess(unittest.TestCase): def test_simple(self): log = self.run_script('simple.py') self.assertIn('Spider closed (finished)', log) - self.assertNotIn("DEBUG: Asyncio support enabled", log) + self.assertNotIn("DEBUG: Asyncio reactor is installed", log) def test_asyncio_enabled_no_reactor(self): log = self.run_script('asyncio_enabled_no_reactor.py') self.assertIn('Spider closed (finished)', log) - self.assertIn("DEBUG: Asyncio support enabled", log) + self.assertIn("DEBUG: Asyncio reactor is installed", log) def test_asyncio_enabled_reactor(self): log = self.run_script('asyncio_enabled_reactor.py') self.assertIn('Spider closed (finished)', log) - self.assertIn("DEBUG: Asyncio support enabled", log) + self.assertIn("DEBUG: Asyncio reactor is installed", log) From 82861c73c8f74c3416f5fa5d77ded33802ca535f Mon Sep 17 00:00:00 2001 From: Atul Gopinathan <41539794+atul-g@users.noreply.github.com> Date: Fri, 27 Dec 2019 22:57:58 +0530 Subject: [PATCH 112/313] Edited the link of the homepage of lxml website The link "https://lxml.de" is redirecting to a completely different and unintended website. I changed the link to the index page of lxml's official website. I thought of changing it to the PyPi page of lxml, but even they are providing the same "https://lxml.de" link which doesn't seem to be working now. --- docs/intro/install.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 51b41b4d7..8ce13d715 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -278,7 +278,7 @@ For details, see `Issue #2473 <https://github.com/scrapy/scrapy/issues/2473>`_. .. _Python: https://www.python.org/ .. _pip: https://pip.pypa.io/en/latest/installing/ -.. _lxml: http://lxml.de/ +.. _lxml: https://lxml.de/index.html .. _parsel: https://pypi.python.org/pypi/parsel .. _w3lib: https://pypi.python.org/pypi/w3lib .. _twisted: https://twistedmatrix.com/ From 14d4428e705e9ca739a4fd041ce7e3134749363c Mon Sep 17 00:00:00 2001 From: 1um0s <abitha95@gmail.com> Date: Mon, 30 Dec 2019 01:26:22 +0530 Subject: [PATCH 113/313] Rephrasing documentation for image and file pipelines (#4252) * scrapy#4034 Clarify documentation for image and file pipelines * scrapy#4034 Clarify documentation for file pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Clarify documentation for image and file pipelines * scrapy#4034 Clarify documentation for file pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Revert image, file pipeline docs. Enhance custom media pipeline docs. * scrapy#4034 rebase master * scrapy#4034 Clarify documentation for image and file pipelines * scrapy#4034 Clarify documentation for file pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Clarify documentation for image and file pipelines * scrapy#4034 Clarify documentation for file pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Simplify documentation for pipeline * scrapy#4034 Revert image, file pipeline docs. Enhance custom media pipeline docs. * scrapy#4034 rebase master * Rebase master * Add class to media pipeline docs Co-Authored-By: elacuesta <elacuesta@users.noreply.github.com> Co-authored-by: elacuesta <elacuesta@users.noreply.github.com> --- docs/topics/media-pipeline.rst | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index a40682e5b..332a14eb7 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -97,7 +97,6 @@ For Files Pipeline, use:: ITEM_PIPELINES = {'scrapy.pipelines.files.FilesPipeline': 1} - .. note:: You can also use both the Files and Images Pipeline at the same time. @@ -578,4 +577,12 @@ above:: item['image_paths'] = image_paths return item + +To enable your custom media pipeline component you must add its class import path to the +:setting:`ITEM_PIPELINES` setting, like in the following example:: + + ITEM_PIPELINES = { + 'myproject.pipelines.MyImagesPipeline': 300 + } + .. _MD5 hash: https://en.wikipedia.org/wiki/MD5 From 21f50c795ac6978de9e73b64f31542a9df928ac5 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 30 Jul 2019 19:46:18 +0500 Subject: [PATCH 114/313] Add async def support to downloader middlewares. --- scrapy/core/downloader/middleware.py | 8 ++++---- tests/test_downloadermiddleware.py | 24 ++++++++++++++++++++++++ 2 files changed, 28 insertions(+), 4 deletions(-) diff --git a/scrapy/core/downloader/middleware.py b/scrapy/core/downloader/middleware.py index 38608a429..9c0014206 100644 --- a/scrapy/core/downloader/middleware.py +++ b/scrapy/core/downloader/middleware.py @@ -8,7 +8,7 @@ from twisted.internet import defer from scrapy.exceptions import _InvalidOutput from scrapy.http import Request, Response from scrapy.middleware import MiddlewareManager -from scrapy.utils.defer import mustbe_deferred +from scrapy.utils.defer import mustbe_deferred, deferred_from_coro from scrapy.utils.conf import build_component_list @@ -33,7 +33,7 @@ class DownloaderMiddlewareManager(MiddlewareManager): @defer.inlineCallbacks def process_request(request): for method in self.methods['process_request']: - response = yield method(request=request, spider=spider) + response = yield deferred_from_coro(method(request=request, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): raise _InvalidOutput('Middleware %s.process_request must return None, Response or Request, got %s' % \ (method.__self__.__class__.__name__, response.__class__.__name__)) @@ -48,7 +48,7 @@ class DownloaderMiddlewareManager(MiddlewareManager): defer.returnValue(response) for method in self.methods['process_response']: - response = yield method(request=request, response=response, spider=spider) + response = yield deferred_from_coro(method(request=request, response=response, spider=spider)) if not isinstance(response, (Response, Request)): raise _InvalidOutput('Middleware %s.process_response must return Response or Request, got %s' % \ (method.__self__.__class__.__name__, type(response))) @@ -60,7 +60,7 @@ class DownloaderMiddlewareManager(MiddlewareManager): def process_exception(_failure): exception = _failure.value for method in self.methods['process_exception']: - response = yield method(request=request, exception=exception, spider=spider) + response = yield deferred_from_coro(method(request=request, exception=exception, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): raise _InvalidOutput('Middleware %s.process_exception must return None, Response or Request, got %s' % \ (method.__self__.__class__.__name__, type(response))) diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 1b81ea949..135321d0a 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -1,3 +1,4 @@ +import asyncio from unittest import mock from twisted.internet.defer import Deferred @@ -206,3 +207,26 @@ class MiddlewareUsingDeferreds(ManagerTestCase): self.assertIs(results[0], resp) self.assertFalse(download_func.called) + + +class MiddlewareUsingCoro(ManagerTestCase): + """Middlewares using asyncio coroutines should work""" + + def test_asyncdef(self): + resp = Response('http://example.com/index.html') + + class CoroMiddleware: + async def process_request(self, request, spider): + await asyncio.sleep(0.1) + return resp + + self.mwman._add_middleware(CoroMiddleware()) + req = Request('http://example.com/index.html') + download_func = mock.MagicMock() + dfd = self.mwman.download(download_func, req, self.spider) + results = [] + dfd.addBoth(results.append) + self._wait(dfd) + + self.assertIs(results[0], resp) + self.assertFalse(download_func.called) From 3603644552f8d15d203abca221edb56047119528 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 12 Sep 2019 20:25:29 +0500 Subject: [PATCH 115/313] Add a non-asyncio async def middleware test. --- tests/test_downloadermiddleware.py | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 135321d0a..5b0cf1eb7 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -1,6 +1,7 @@ import asyncio from unittest import mock +from twisted.internet import defer from twisted.internet.defer import Deferred from twisted.trial.unittest import TestCase from twisted.python.failure import Failure @@ -215,6 +216,25 @@ class MiddlewareUsingCoro(ManagerTestCase): def test_asyncdef(self): resp = Response('http://example.com/index.html') + class CoroMiddleware: + async def process_request(self, request, spider): + await defer.succeed(42) + return resp + + self.mwman._add_middleware(CoroMiddleware()) + req = Request('http://example.com/index.html') + download_func = mock.MagicMock() + dfd = self.mwman.download(download_func, req, self.spider) + results = [] + dfd.addBoth(results.append) + self._wait(dfd) + + self.assertIs(results[0], resp) + self.assertFalse(download_func.called) + + def test_asyncdef_asyncio(self): + resp = Response('http://example.com/index.html') + class CoroMiddleware: async def process_request(self, request, spider): await asyncio.sleep(0.1) From 5cf1ac0005fa3a174b700d88c0e2536b689f13c7 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 19:24:44 +0500 Subject: [PATCH 116/313] Move the asyncio downloader mw test to a separate class. --- tests/test_downloadermiddleware.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 5b0cf1eb7..c5c4d13bd 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -1,6 +1,7 @@ import asyncio from unittest import mock +from pytest import mark from twisted.internet import defer from twisted.internet.defer import Deferred from twisted.trial.unittest import TestCase @@ -232,6 +233,12 @@ class MiddlewareUsingCoro(ManagerTestCase): self.assertIs(results[0], resp) self.assertFalse(download_func.called) + +@mark.only_asyncio() +class MiddlewareUsingCoroAsyncio(ManagerTestCase): + + settings_dict = {'ASYNCIO_ENABLED': True} + def test_asyncdef_asyncio(self): resp = Response('http://example.com/index.html') From 50aa6ef22cc9dec3655c9e3aaee225f159e94df1 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Sat, 21 Dec 2019 14:36:11 +0500 Subject: [PATCH 117/313] Add deferred_from_coro. --- scrapy/utils/defer.py | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 20ce59297..bbd5ebe52 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -1,10 +1,14 @@ """ Helper functions for dealing with Twisted deferreds """ +import asyncio +import inspect + from twisted.internet import defer, task from twisted.python import failure from scrapy.exceptions import IgnoreRequest +from scrapy.utils.asyncio import is_asyncio_reactor_installed def defer_fail(_failure): @@ -114,3 +118,25 @@ def iter_errback(iterable, errback, *a, **kw): break except Exception: errback(failure.Failure(), *a, **kw) + + +def _isfuture(o): + # workaround for Python before 3.5.3 not having asyncio.isfuture + if hasattr(asyncio, 'isfuture'): + return asyncio.isfuture(o) + return isinstance(o, asyncio.Future) + + +def deferred_from_coro(o): + """Converts a coroutine into a Deferred, or returns the object as is if it isn't a coroutine""" + if isinstance(o, defer.Deferred): + return o + if _isfuture(o) or inspect.isawaitable(o): + if not is_asyncio_reactor_installed(): + # wrapping the coroutine directly into a Deferred, this doesn't work correctly with coroutines + # that use asyncio, e.g. "await asyncio.sleep(1)" + return defer.ensureDeferred(o) + else: + # wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor + return defer.Deferred.fromFuture(asyncio.ensure_future(o)) + return o From 16787f5bf4475fe1604c1e4cac7b491f1df6a1fb Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 30 Dec 2019 12:02:19 +0500 Subject: [PATCH 118/313] Merge middleware tests back as we don't need to set the setting anymore. --- tests/test_downloadermiddleware.py | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index c5c4d13bd..3943cecf7 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -233,12 +233,7 @@ class MiddlewareUsingCoro(ManagerTestCase): self.assertIs(results[0], resp) self.assertFalse(download_func.called) - -@mark.only_asyncio() -class MiddlewareUsingCoroAsyncio(ManagerTestCase): - - settings_dict = {'ASYNCIO_ENABLED': True} - + @mark.only_asyncio() def test_asyncdef_asyncio(self): resp = Response('http://example.com/index.html') From e3b8ba6188ce703e43563d8588b7d709b037a0e0 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 31 Dec 2019 17:54:01 +0500 Subject: [PATCH 119/313] Run py35-asyncio also on 3.5.2 to test Xenial. --- .travis.yml | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.travis.yml b/.travis.yml index 82167e10a..c808b3436 100644 --- a/.travis.yml +++ b/.travis.yml @@ -18,6 +18,8 @@ matrix: python: 3.5 - env: TOXENV=py35-asyncio python: 3.5 + - env: TOXENV=py35-asyncio + python: 3.5.2 - env: TOXENV=py36 python: 3.6 - env: TOXENV=py37 From 2b9254c2bde76995e81c54ad112ae884a1499386 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 31 Dec 2019 17:54:41 +0500 Subject: [PATCH 120/313] Add a test function that uses asyncio.Queue(). --- scrapy/utils/test.py | 9 ++++++++- tests/test_downloadermiddleware.py | 5 +++-- 2 files changed, 11 insertions(+), 3 deletions(-) diff --git a/scrapy/utils/test.py b/scrapy/utils/test.py index 307c25352..0f4cf8091 100644 --- a/scrapy/utils/test.py +++ b/scrapy/utils/test.py @@ -1,7 +1,7 @@ """ This module contains some assorted functions used in tests """ - +import asyncio import os from importlib import import_module @@ -96,3 +96,10 @@ def assert_samelines(testcase, text1, text2, msg=None): line endings between platforms """ testcase.assertEqual(text1.splitlines(), text2.splitlines(), msg) + + +def get_from_asyncio_queue(value): + q = asyncio.Queue() + getter = q.get() + q.put_nowait(value) + return getter diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 3943cecf7..3dd4f2351 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -11,7 +11,7 @@ from scrapy.http import Request, Response from scrapy.spiders import Spider from scrapy.exceptions import _InvalidOutput from scrapy.core.downloader.middleware import DownloaderMiddlewareManager -from scrapy.utils.test import get_crawler +from scrapy.utils.test import get_crawler, get_from_asyncio_queue from scrapy.utils.python import to_bytes @@ -240,7 +240,8 @@ class MiddlewareUsingCoro(ManagerTestCase): class CoroMiddleware: async def process_request(self, request, spider): await asyncio.sleep(0.1) - return resp + result = await get_from_asyncio_queue(resp) + return result self.mwman._add_middleware(CoroMiddleware()) req = Request('http://example.com/index.html') From b2dd379bc2b93b4863dd36a296780481d758c4cb Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 3 Jan 2020 21:38:05 +0500 Subject: [PATCH 121/313] Remove the py35-asyncio env for 3.5 from Travis. --- .travis.yml | 2 -- 1 file changed, 2 deletions(-) diff --git a/.travis.yml b/.travis.yml index c808b3436..66e1a9617 100644 --- a/.travis.yml +++ b/.travis.yml @@ -16,8 +16,6 @@ matrix: python: 3.5 - env: TOXENV=pinned python: 3.5 - - env: TOXENV=py35-asyncio - python: 3.5 - env: TOXENV=py35-asyncio python: 3.5.2 - env: TOXENV=py36 From 81175669746737f1ab0bd0236a70f787ed1855a4 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 22:12:27 +0500 Subject: [PATCH 122/313] Add utils.defer.deferred_f_from_coro_f. --- scrapy/utils/defer.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index bbd5ebe52..62b43a96c 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -2,6 +2,7 @@ Helper functions for dealing with Twisted deferreds """ import asyncio +from functools import wraps import inspect from twisted.internet import defer, task @@ -140,3 +141,15 @@ def deferred_from_coro(o): # wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor return defer.Deferred.fromFuture(asyncio.ensure_future(o)) return o + + +def deferred_f_from_coro_f(coro_f): + """ Converts a coroutine function into a function that returns a Deferred. + + The coroutine function will be called at the time when the wrapper is called. Wrapper args will be passed to it. + This is useful for callback chains, as callback functions are called with the previous callback result. + """ + @wraps(coro_f) + def f(*coro_args, **coro_kwargs): + return deferred_from_coro(coro_f(*coro_args, **coro_kwargs)) + return f From 1f9cef787d3ca0c12099f1b1b4c52efc510e381d Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 10 Sep 2019 14:26:21 +0500 Subject: [PATCH 123/313] Add async def support to pipelines. --- scrapy/pipelines/__init__.py | 4 +++- tests/test_pipelines.py | 13 +++++++++++++ 2 files changed, 16 insertions(+), 1 deletion(-) diff --git a/scrapy/pipelines/__init__.py b/scrapy/pipelines/__init__.py index aa1bfb77f..1a45e00a2 100644 --- a/scrapy/pipelines/__init__.py +++ b/scrapy/pipelines/__init__.py @@ -6,6 +6,8 @@ See documentation in docs/item-pipeline.rst from scrapy.middleware import MiddlewareManager from scrapy.utils.conf import build_component_list +from scrapy.utils.defer import deferred_f_from_coro_f + class ItemPipelineManager(MiddlewareManager): @@ -19,7 +21,7 @@ class ItemPipelineManager(MiddlewareManager): def _add_middleware(self, pipe): super(ItemPipelineManager, self)._add_middleware(pipe) if hasattr(pipe, 'process_item'): - self.methods['process_item'].append(pipe.process_item) + self.methods['process_item'].append(deferred_f_from_coro_f(pipe.process_item)) def process_item(self, item, spider): return self._process_chain('process_item', item, spider) diff --git a/tests/test_pipelines.py b/tests/test_pipelines.py index bc53f5427..cfe4471d7 100644 --- a/tests/test_pipelines.py +++ b/tests/test_pipelines.py @@ -26,6 +26,13 @@ class DeferredPipeline: return d +class AsyncDefPipeline: + async def process_item(self, item, spider): + await defer.succeed(42) + item['pipeline_passed'] = True + return item + + class ItemSpider(Spider): name = 'itemspider' @@ -69,3 +76,9 @@ class PipelineTestCase(unittest.TestCase): crawler = self._create_crawler(DeferredPipeline) yield crawler.crawl(mockserver=self.mockserver) self.assertEqual(len(self.items), 1) + + @defer.inlineCallbacks + def test_asyncdef_pipeline(self): + crawler = self._create_crawler(AsyncDefPipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) From bfdd552a32b3a67dfa2895b483c187d87b63b50e Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 10 Sep 2019 14:57:07 +0500 Subject: [PATCH 124/313] Add a test for pipelines using asyncio. --- tests/test_pipelines.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/tests/test_pipelines.py b/tests/test_pipelines.py index cfe4471d7..aba6d85b7 100644 --- a/tests/test_pipelines.py +++ b/tests/test_pipelines.py @@ -1,3 +1,5 @@ +import asyncio + from twisted.internet import defer from twisted.internet.defer import Deferred from twisted.trial import unittest @@ -33,6 +35,13 @@ class AsyncDefPipeline: return item +class AsyncDefAsyncioPipeline: + async def process_item(self, item, spider): + await asyncio.sleep(0.2) + item['pipeline_passed'] = True + return item + + class ItemSpider(Spider): name = 'itemspider' @@ -82,3 +91,9 @@ class PipelineTestCase(unittest.TestCase): crawler = self._create_crawler(AsyncDefPipeline) yield crawler.crawl(mockserver=self.mockserver) self.assertEqual(len(self.items), 1) + + @defer.inlineCallbacks + def test_asyncdef_asyncio_pipeline(self): + crawler = self._create_crawler(AsyncDefAsyncioPipeline) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 1) From bdef948aaebced3d28ea7b81525ea9edf7cc650c Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 23:15:43 +0500 Subject: [PATCH 125/313] Mark the asyncio pipelines test as only_asyncio. --- tests/test_pipelines.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/test_pipelines.py b/tests/test_pipelines.py index aba6d85b7..6f33282f2 100644 --- a/tests/test_pipelines.py +++ b/tests/test_pipelines.py @@ -1,5 +1,6 @@ import asyncio +from pytest import mark from twisted.internet import defer from twisted.internet.defer import Deferred from twisted.trial import unittest @@ -92,6 +93,7 @@ class PipelineTestCase(unittest.TestCase): yield crawler.crawl(mockserver=self.mockserver) self.assertEqual(len(self.items), 1) + @mark.only_asyncio() @defer.inlineCallbacks def test_asyncdef_asyncio_pipeline(self): crawler = self._create_crawler(AsyncDefAsyncioPipeline) From 9d8c54c0f2a98dc627efe8a2b58cfd311f2105cb Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Mon, 16 Dec 2019 22:43:55 +0500 Subject: [PATCH 126/313] Fix/ignore flake8 problems. --- pytest.ini | 1 + scrapy/pipelines/__init__.py | 1 - 2 files changed, 1 insertion(+), 1 deletion(-) diff --git a/pytest.ini b/pytest.ini index c3f3292bb..1030d5530 100644 --- a/pytest.ini +++ b/pytest.ini @@ -95,6 +95,7 @@ flake8-ignore = scrapy/loader/__init__.py E501 E128 scrapy/loader/processors.py E501 # scrapy/pipelines + scrapy/pipelines/__init__.py E501 scrapy/pipelines/files.py E116 E501 E266 scrapy/pipelines/images.py E265 E501 scrapy/pipelines/media.py E125 E501 E266 diff --git a/scrapy/pipelines/__init__.py b/scrapy/pipelines/__init__.py index 1a45e00a2..b5725a8ee 100644 --- a/scrapy/pipelines/__init__.py +++ b/scrapy/pipelines/__init__.py @@ -9,7 +9,6 @@ from scrapy.utils.conf import build_component_list from scrapy.utils.defer import deferred_f_from_coro_f - class ItemPipelineManager(MiddlewareManager): component_name = 'item pipeline' From 7d859848800ef05760e4f57dc206a4d756460a9d Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 9 Jan 2020 14:48:07 +0500 Subject: [PATCH 127/313] Use get_from_asyncio_queue in the pipeline test. --- tests/test_pipelines.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_pipelines.py b/tests/test_pipelines.py index 6f33282f2..c72f1a338 100644 --- a/tests/test_pipelines.py +++ b/tests/test_pipelines.py @@ -6,7 +6,7 @@ from twisted.internet.defer import Deferred from twisted.trial import unittest from scrapy import Spider, signals, Request -from scrapy.utils.test import get_crawler +from scrapy.utils.test import get_crawler, get_from_asyncio_queue from tests.mockserver import MockServer @@ -39,7 +39,7 @@ class AsyncDefPipeline: class AsyncDefAsyncioPipeline: async def process_item(self, item, spider): await asyncio.sleep(0.2) - item['pipeline_passed'] = True + item['pipeline_passed'] = await get_from_asyncio_queue(True) return item From 3faef2d08277f43659400470ec408ee29400018a Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 12 Sep 2019 20:10:58 +0500 Subject: [PATCH 128/313] Add async def support to signal handlers that already supported Deferreds. --- scrapy/utils/defer.py | 17 +++++++++++++++++ scrapy/utils/signal.py | 6 ++++-- tests/test_utils_signal.py | 26 +++++++++++++++++++++++--- 3 files changed, 44 insertions(+), 5 deletions(-) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index bbd5ebe52..a2c24e5fb 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -140,3 +140,20 @@ def deferred_from_coro(o): # wrapping the coroutine into a Future and then into a Deferred, this requires AsyncioSelectorReactor return defer.Deferred.fromFuture(asyncio.ensure_future(o)) return o + + +def maybeDeferred_coro(f, *args, **kw): + """ Copy of defer.maybeDeferred that also converts coroutines to Deferreds. """ + try: + result = f(*args, **kw) + except: # noqa: E722 + return defer.fail(failure.Failure(captureVars=defer.Deferred.debug)) + + if isinstance(result, defer.Deferred): + return result + elif _isfuture(result) or inspect.isawaitable(result): + return deferred_from_coro(result) + elif isinstance(result, failure.Failure): + return defer.fail(result) + else: + return defer.succeed(result) diff --git a/scrapy/utils/signal.py b/scrapy/utils/signal.py index de00bac49..60c561da6 100644 --- a/scrapy/utils/signal.py +++ b/scrapy/utils/signal.py @@ -2,12 +2,14 @@ import logging -from twisted.internet.defer import maybeDeferred, DeferredList, Deferred +from twisted.internet.defer import DeferredList, Deferred from twisted.python.failure import Failure from pydispatch.dispatcher import Any, Anonymous, liveReceivers, \ getAllReceivers, disconnect from pydispatch.robustapply import robustApply + +from scrapy.utils.defer import maybeDeferred_coro from scrapy.utils.log import failure_to_exc_info logger = logging.getLogger(__name__) @@ -61,7 +63,7 @@ def send_catch_log_deferred(signal=Any, sender=Anonymous, *arguments, **named): spider = named.get('spider', None) dfds = [] for receiver in liveReceivers(getAllReceivers(sender, signal)): - d = maybeDeferred(robustApply, receiver, signal=signal, sender=sender, + d = maybeDeferred_coro(robustApply, receiver, signal=signal, sender=sender, *arguments, **named) d.addErrback(logerror, receiver) d.addBoth(lambda result: (receiver, result)) diff --git a/tests/test_utils_signal.py b/tests/test_utils_signal.py index 16b7c5c68..e5f6f0ed4 100644 --- a/tests/test_utils_signal.py +++ b/tests/test_utils_signal.py @@ -1,3 +1,6 @@ +import asyncio + +from pytest import mark from testfixtures import LogCapture from twisted.trial import unittest from twisted.python.failure import Failure @@ -5,6 +8,7 @@ from twisted.internet import defer, reactor from pydispatch import dispatcher from scrapy.utils.signal import send_catch_log, send_catch_log_deferred +from scrapy.utils.test import get_from_asyncio_queue class SendCatchLogTest(unittest.TestCase): @@ -54,7 +58,7 @@ class SendCatchLogDeferredTest(SendCatchLogTest): return send_catch_log_deferred(signal, *a, **kw) -class SendCatchLogDeferredTest2(SendCatchLogTest): +class SendCatchLogDeferredTest2(SendCatchLogDeferredTest): def ok_handler(self, arg, handlers_called): handlers_called.add(self.ok_handler) @@ -63,8 +67,24 @@ class SendCatchLogDeferredTest2(SendCatchLogTest): reactor.callLater(0, d.callback, "OK") return d - def _get_result(self, signal, *a, **kw): - return send_catch_log_deferred(signal, *a, **kw) + +class SendCatchLogDeferredAsyncDefTest(SendCatchLogDeferredTest): + + async def ok_handler(self, arg, handlers_called): + handlers_called.add(self.ok_handler) + assert arg == 'test' + await defer.succeed(42) + return "OK" + + +@mark.only_asyncio() +class SendCatchLogDeferredAsyncioTest(SendCatchLogDeferredTest): + + async def ok_handler(self, arg, handlers_called): + handlers_called.add(self.ok_handler) + assert arg == 'test' + await asyncio.sleep(0.2) + return await get_from_asyncio_queue("OK") class SendCatchLogTest2(unittest.TestCase): From a91a13b4434ae60be6fded94e4ed08ba322b4f69 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 12 Nov 2019 23:09:00 +0500 Subject: [PATCH 129/313] Support for async def callbacks. --- scrapy/utils/spider.py | 5 +++-- tests/spiders.py | 23 +++++++++++++++++++++++ tests/test_crawl.py | 20 +++++++++++++++++++- 3 files changed, 45 insertions(+), 3 deletions(-) diff --git a/scrapy/utils/spider.py b/scrapy/utils/spider.py index 4061d1ea3..72775df5c 100644 --- a/scrapy/utils/spider.py +++ b/scrapy/utils/spider.py @@ -2,14 +2,15 @@ import logging import inspect from scrapy.spiders import Spider -from scrapy.utils.misc import arg_to_iter +from scrapy.utils.defer import deferred_from_coro +from scrapy.utils.misc import arg_to_iter logger = logging.getLogger(__name__) def iterate_spider_output(result): - return arg_to_iter(result) + return arg_to_iter(deferred_from_coro(result)) def iter_spider_classes(module): diff --git a/tests/spiders.py b/tests/spiders.py index 39c8da0b6..e4f2d5474 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -1,14 +1,18 @@ """ Some spiders used for testing and benchmarking """ +import asyncio import time from urllib.parse import urlencode +from twisted.internet import defer + from scrapy.http import Request from scrapy.item import Item from scrapy.linkextractors import LinkExtractor from scrapy.spiders import Spider from scrapy.spiders.crawl import CrawlSpider, Rule +from scrapy.utils.test import get_from_asyncio_queue class MockServerSpider(Spider): @@ -83,6 +87,25 @@ class SimpleSpider(MetaSpider): self.logger.info("Got response %d" % response.status) +class AsyncDefSpider(SimpleSpider): + + name = 'asyncdef' + + async def parse(self, response): + await defer.succeed(42) + self.logger.info("Got response %d" % response.status) + + +class AsyncDefAsyncioSpider(SimpleSpider): + + name = 'asyncdef_asyncio' + + async def parse(self, response): + await asyncio.sleep(0.2) + status = await get_from_asyncio_queue(response.status) + self.logger.info("Got response %d" % status) + + class ItemSpider(FollowAllSpider): name = 'item' diff --git a/tests/test_crawl.py b/tests/test_crawl.py index f433fcea6..99b887ff6 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -1,6 +1,7 @@ import json import logging +from pytest import mark from testfixtures import LogCapture from twisted.internet import defer from twisted.trial.unittest import TestCase @@ -10,7 +11,8 @@ from scrapy.http import Request from scrapy.utils.python import to_unicode from tests.mockserver import MockServer from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider, - SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback) + SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback, + AsyncDefSpider, AsyncDefAsyncioSpider) class CrawlTestCase(TestCase): @@ -308,3 +310,19 @@ with multiples lines self.assertIn("[callback] status 201", str(log)) self.assertIn("[errback] status 404", str(log)) self.assertIn("[errback] status 500", str(log)) + + @defer.inlineCallbacks + def test_async_def_parse(self): + self.runner.crawl(AsyncDefSpider, self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + with LogCapture() as log: + yield self.runner.join() + self.assertIn("Got response 200", str(log)) + + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncio_parse(self): + runner = CrawlerRunner({"ASYNCIO_REACTOR": True}) + runner.crawl(AsyncDefAsyncioSpider, self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + with LogCapture() as log: + yield runner.join() + self.assertIn("Got response 200", str(log)) From 6ce1ad31071326d22387f8444abefd6f3e18ed86 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Jan 2020 04:20:37 -0300 Subject: [PATCH 130/313] [test] Spider middleware: catch exceptions right after the spider callback --- tests/test_spidermiddleware_output_chain.py | 50 +++++++++++++++++++-- 1 file changed, 46 insertions(+), 4 deletions(-) diff --git a/tests/test_spidermiddleware_output_chain.py b/tests/test_spidermiddleware_output_chain.py index 739cf1c2d..b19a74609 100644 --- a/tests/test_spidermiddleware_output_chain.py +++ b/tests/test_spidermiddleware_output_chain.py @@ -1,10 +1,10 @@ - from testfixtures import LogCapture -from twisted.trial.unittest import TestCase from twisted.internet import defer +from twisted.trial.unittest import TestCase -from scrapy import Spider, Request +from scrapy import Request, Spider from scrapy.utils.test import get_crawler + from tests.mockserver import MockServer @@ -74,7 +74,7 @@ class ProcessSpiderInputSpiderWithErrback(ProcessSpiderInputSpiderWithoutErrback name = 'ProcessSpiderInputSpiderWithErrback' def start_requests(self): - yield Request(url=self.mockserver.url('/status?n=200'), callback=self.parse, errback=self.errback) + yield Request(self.mockserver.url('/status?n=200'), self.parse, errback=self.errback) def errback(self, failure): self.logger.info('Got a Failure on the Request errback') @@ -100,6 +100,17 @@ class GeneratorCallbackSpider(Spider): raise ImportError() +# ================================================================================ +# (2.1) exceptions from a spider callback (generator, middleware right after callback) +class GeneratorCallbackSpiderMiddlewareRightAfterSpider(GeneratorCallbackSpider): + name = 'GeneratorCallbackSpiderMiddlewareRightAfterSpider' + custom_settings = { + 'SPIDER_MIDDLEWARES': { + __name__ + '.LogExceptionMiddleware': 100000, + }, + } + + # ================================================================================ # (3) exceptions from a spider callback (not a generator) class NotGeneratorCallbackSpider(Spider): @@ -117,6 +128,17 @@ class NotGeneratorCallbackSpider(Spider): return [{'test': 1}, {'test': 1/0}] +# ================================================================================ +# (3.1) exceptions from a spider callback (not a generator, middleware right after callback) +class NotGeneratorCallbackSpiderMiddlewareRightAfterSpider(NotGeneratorCallbackSpider): + name = 'NotGeneratorCallbackSpiderMiddlewareRightAfterSpider' + custom_settings = { + 'SPIDER_MIDDLEWARES': { + __name__ + '.LogExceptionMiddleware': 100000, + }, + } + + # ================================================================================ # (4) exceptions from a middleware process_spider_output method (generator) class GeneratorOutputChainSpider(Spider): @@ -320,6 +342,16 @@ class TestSpiderMiddleware(TestCase): self.assertIn("Middleware: ImportError exception caught", str(log2)) self.assertIn("'item_scraped_count': 2", str(log2)) + @defer.inlineCallbacks + def test_generator_callback_right_after_callback(self): + """ + (2.1) Special case of (2): Exceptions should be caught + even if the middleware is placed right after the spider + """ + log21 = yield self.crawl_log(GeneratorCallbackSpiderMiddlewareRightAfterSpider) + self.assertIn("Middleware: ImportError exception caught", str(log21)) + self.assertIn("'item_scraped_count': 2", str(log21)) + @defer.inlineCallbacks def test_not_a_generator_callback(self): """ @@ -330,6 +362,16 @@ class TestSpiderMiddleware(TestCase): self.assertIn("Middleware: ZeroDivisionError exception caught", str(log3)) self.assertNotIn("item_scraped_count", str(log3)) + @defer.inlineCallbacks + def test_not_a_generator_callback_right_after_callback(self): + """ + (3.1) Special case of (3): Exceptions should be caught + even if the middleware is placed right after the spider + """ + log31 = yield self.crawl_log(NotGeneratorCallbackSpiderMiddlewareRightAfterSpider) + self.assertIn("Middleware: ZeroDivisionError exception caught", str(log31)) + self.assertNotIn("item_scraped_count", str(log31)) + @defer.inlineCallbacks def test_generator_output_chain(self): """ From c088c04f449c3383b7867c25b387f13949b1d6c0 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Jan 2020 04:20:55 -0300 Subject: [PATCH 131/313] Spider middleware: catch exceptions right after the spider callback --- scrapy/core/spidermw.py | 71 +++++++++++++++++++++++++---------------- scrapy/utils/python.py | 1 + 2 files changed, 44 insertions(+), 28 deletions(-) diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index 097a374bf..180a0b1fe 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -3,13 +3,14 @@ Spider Middleware manager See documentation in docs/topics/spider-middleware.rst """ -from itertools import chain, islice +from itertools import islice from twisted.python.failure import Failure + from scrapy.exceptions import _InvalidOutput from scrapy.middleware import MiddlewareManager -from scrapy.utils.defer import mustbe_deferred from scrapy.utils.conf import build_component_list +from scrapy.utils.defer import mustbe_deferred from scrapy.utils.python import MutableChain @@ -17,6 +18,13 @@ def _isiterable(possible_iterator): return hasattr(possible_iterator, '__iter__') +def _fname(f): + return "%s.%s".format( + f.__self__.__class__.__name__, + f.__func__.__name__ + ) + + class SpiderMiddlewareManager(MiddlewareManager): component_name = 'spider middleware' @@ -31,27 +39,36 @@ class SpiderMiddlewareManager(MiddlewareManager): self.methods['process_spider_input'].append(mw.process_spider_input) if hasattr(mw, 'process_start_requests'): self.methods['process_start_requests'].appendleft(mw.process_start_requests) - self.methods['process_spider_output'].appendleft(getattr(mw, 'process_spider_output', None)) - self.methods['process_spider_exception'].appendleft(getattr(mw, 'process_spider_exception', None)) + process_spider_output = getattr(mw, 'process_spider_output', None) + self.methods['process_spider_output'].appendleft(process_spider_output) + process_spider_exception = getattr(mw, 'process_spider_exception', None) + self.methods['process_spider_exception'].appendleft(process_spider_exception) def scrape_response(self, scrape_func, response, request, spider): - fname = lambda f: '%s.%s' % ( - f.__self__.__class__.__name__, - f.__func__.__name__) def process_spider_input(response): for method in self.methods['process_spider_input']: try: result = method(response=response, spider=spider) if result is not None: - raise _InvalidOutput('Middleware {} must return None or raise an exception, got {}' - .format(fname(method), type(result))) + msg = "Middleware {} must return None or raise an exception, got {}" + raise _InvalidOutput(msg.format(_fname(method), type(result))) except _InvalidOutput: raise except Exception: return scrape_func(Failure(), request, spider) return scrape_func(response, request, spider) + def _evaluate_iterable(iterable, method_index, recover_to): + try: + for r in iterable: + yield r + except Exception as ex: + exception_result = process_spider_exception(Failure(ex), method_index) + if isinstance(exception_result, Failure): + raise + recover_to.extend(exception_result) + def process_spider_exception(_failure, start_index=0): exception = _failure.value # don't handle _InvalidOutput exception @@ -69,8 +86,8 @@ class SpiderMiddlewareManager(MiddlewareManager): elif result is None: continue else: - raise _InvalidOutput('Middleware {} must return None or an iterable, got {}' - .format(fname(method), type(result))) + msg = "Middleware {} must return None or an iterable, got {}" + raise _InvalidOutput(msg.format(_fname(method), type(result))) return _failure def process_spider_output(result, start_index=0): @@ -78,38 +95,36 @@ class SpiderMiddlewareManager(MiddlewareManager): # chain, they went through it already from the process_spider_exception method recovered = MutableChain() - def evaluate_iterable(iterable, index): - try: - for r in iterable: - yield r - except Exception as ex: - exception_result = process_spider_exception(Failure(ex), index+1) - if isinstance(exception_result, Failure): - raise - recovered.extend(exception_result) - method_list = islice(self.methods['process_spider_output'], start_index, None) for method_index, method in enumerate(method_list, start=start_index): if method is None: continue - # the following might fail directly if the output value is not a generator try: + # might fail directly if the output value is not a generator result = method(response=response, result=result, spider=spider) except Exception as ex: exception_result = process_spider_exception(Failure(ex), method_index+1) if isinstance(exception_result, Failure): raise return exception_result - if _isiterable(result): - result = evaluate_iterable(result, method_index) else: - raise _InvalidOutput('Middleware {} must return an iterable, got {}' - .format(fname(method), type(result))) + if _isiterable(result): + result = _evaluate_iterable(result, method_index+1, recovered) + else: + msg = "Middleware {} must return an iterable, got {}" + raise _InvalidOutput(msg.format(_fname(method), type(result))) - return chain(result, recovered) + return MutableChain(result, recovered) + + def process_callback_output(result): + if isinstance(result, Failure): + return process_spider_exception(result) + recovered = MutableChain() + result = _evaluate_iterable(result, 0, recovered) + return MutableChain(process_spider_output(result), recovered) dfd = mustbe_deferred(process_spider_input, response) - dfd.addCallbacks(callback=process_spider_output, errback=process_spider_exception) + dfd.addCallbacks(callback=process_callback_output, errback=process_callback_output) return dfd def process_start_requests(self, start_requests, spider): diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 8d829c5a5..875650f3e 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -375,6 +375,7 @@ class MutableChain(object): """ Thin wrapper around itertools.chain, allowing to add iterables "in-place" """ + def __init__(self, *args): self.data = chain(*args) From d6e928f47209396d0f6c0b155eb07115c4855c93 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Jan 2020 04:40:03 -0300 Subject: [PATCH 132/313] Remove object as base class for MutableChain Plus some minor styling adjustments --- scrapy/utils/python.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 875650f3e..e5582cc18 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -1,15 +1,15 @@ """ This module contains essential stuff that should've come with Python itself ;) """ +import errno import gc +import inspect import os import re -import inspect +import sys import weakref -import errno from functools import partial, wraps from itertools import chain -import sys from scrapy.utils.decorators import deprecated @@ -371,7 +371,7 @@ else: gc.collect() -class MutableChain(object): +class MutableChain: """ Thin wrapper around itertools.chain, allowing to add iterables "in-place" """ From 9770ca35fb494503298270902569ed897a365d32 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Jan 2020 18:45:39 -0300 Subject: [PATCH 133/313] Spider middleware: simplify deferred errback handling --- scrapy/core/spidermw.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index 180a0b1fe..ed02b306b 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -117,14 +117,12 @@ class SpiderMiddlewareManager(MiddlewareManager): return MutableChain(result, recovered) def process_callback_output(result): - if isinstance(result, Failure): - return process_spider_exception(result) recovered = MutableChain() result = _evaluate_iterable(result, 0, recovered) return MutableChain(process_spider_output(result), recovered) dfd = mustbe_deferred(process_spider_input, response) - dfd.addCallbacks(callback=process_callback_output, errback=process_callback_output) + dfd.addCallbacks(callback=process_callback_output, errback=process_spider_exception) return dfd def process_start_requests(self, start_requests, spider): From 03241aa4a66f9b0e8be4dc104807011f546840e8 Mon Sep 17 00:00:00 2001 From: abhishekh2001 <53903855+abhishekh2001@users.noreply.github.com> Date: Wed, 15 Jan 2020 08:54:25 +0400 Subject: [PATCH 134/313] Fixed artwork/README formatting --- artwork/README.rst | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/artwork/README.rst b/artwork/README.rst index 92f6ecb7e..8a1028cde 100644 --- a/artwork/README.rst +++ b/artwork/README.rst @@ -1,5 +1,4 @@ -:orphan: - +============== Scrapy artwork ============== From 735c0ceb7890dbea607dd6e4c0a14a4ce2b0afd2 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 9 Sep 2019 16:20:58 -0300 Subject: [PATCH 135/313] Custom name resolver implementing twisted.internet.interfaces.IHostnameResolver --- pytest.ini | 1 + scrapy/crawler.py | 38 ++++++++++++++++++---------------- scrapy/resolver.py | 51 ++++++++++++++++++++++++++-------------------- 3 files changed, 50 insertions(+), 40 deletions(-) diff --git a/pytest.ini b/pytest.ini index c3f3292bb..a0e89f0a9 100644 --- a/pytest.ini +++ b/pytest.ini @@ -158,6 +158,7 @@ flake8-ignore = scrapy/mail.py E402 E128 E501 E502 scrapy/middleware.py E128 E501 scrapy/pqueues.py E501 + scrapy/resolver.py E501 scrapy/responsetypes.py E128 E501 E305 scrapy/robotstxt.py E501 scrapy/shell.py E501 diff --git a/scrapy/crawler.py b/scrapy/crawler.py index f87e67d93..1350ea84f 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -4,34 +4,36 @@ import signal import warnings from twisted.internet import defer -from zope.interface.verify import verifyClass, DoesNotImplement +from zope.interface.verify import DoesNotImplement, verifyClass -from scrapy import Spider +from scrapy import signals, Spider from scrapy.core.engine import ExecutionEngine -from scrapy.resolver import CachingThreadedResolver -from scrapy.interfaces import ISpiderLoader +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.extension import ExtensionManager +from scrapy.interfaces import ISpiderLoader +from scrapy.resolver import CachingHostnameResolver from scrapy.settings import overridden_settings, Settings from scrapy.signalmanager import SignalManager -from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_reactor_installed -from scrapy.utils.ossignal import install_shutdown_handlers, signal_names -from scrapy.utils.misc import load_object from scrapy.utils.log import ( - LogCounterHandler, configure_logging, log_scrapy_info, - get_scrapy_root_handler, install_scrapy_root_handler) -from scrapy import signals + configure_logging, + get_scrapy_root_handler, + install_scrapy_root_handler, + log_scrapy_info, + LogCounterHandler, +) +from scrapy.utils.misc import load_object +from scrapy.utils.ossignal import install_shutdown_handlers, signal_names logger = logging.getLogger(__name__) -class Crawler(object): +class Crawler: def __init__(self, spidercls, settings=None): if isinstance(spidercls, Spider): - raise ValueError( - 'The spidercls argument must be a class, not an object') + raise ValueError('The spidercls argument must be a class, not an object') if isinstance(settings, dict) or settings is None: settings = Settings(settings) @@ -110,7 +112,7 @@ class Crawler(object): yield defer.maybeDeferred(self.engine.stop) -class CrawlerRunner(object): +class CrawlerRunner: """ This is a convenient helper class that keeps track of, manages and runs crawlers inside an already setup :mod:`~twisted.internet.reactor`. @@ -303,7 +305,7 @@ class CrawlerProcess(CrawlerRunner): return d.addBoth(self._stop_reactor) - reactor.installResolver(self._get_dns_resolver()) + reactor.installNameResolver(self._get_dns_resolver()) tp = reactor.getThreadPool() tp.adjustPoolsize(maxthreads=self.settings.getint('REACTOR_THREADPOOL_MAXSIZE')) reactor.addSystemEventTrigger('before', 'shutdown', self.stop) @@ -315,10 +317,10 @@ class CrawlerProcess(CrawlerRunner): cache_size = self.settings.getint('DNSCACHE_SIZE') else: cache_size = 0 - return CachingThreadedResolver( - reactor=reactor, + return CachingHostnameResolver( + resolver=reactor.nameResolver, cache_size=cache_size, - timeout=self.settings.getfloat('DNS_TIMEOUT') + timeout=self.settings.getfloat('DNS_TIMEOUT'), ) def _graceful_stop_reactor(self): diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 4df949015..03964f269 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -1,32 +1,39 @@ -from twisted.internet import defer -from twisted.internet.base import ThreadedResolver +from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver +from zope.interface.declarations import implementer, provider from scrapy.utils.datatypes import LocalCache -# TODO: cache misses +# TODO: cache misses dnscache = LocalCache(10000) -class CachingThreadedResolver(ThreadedResolver): - def __init__(self, reactor, cache_size, timeout): - super(CachingThreadedResolver, self).__init__(reactor) - dnscache.limit = cache_size +@implementer(IHostnameResolver) +class CachingHostnameResolver(object): + + def __init__(self, resolver, cache_size, timeout): + self.resolver = resolver self.timeout = timeout + dnscache.limit = cache_size - def getHostByName(self, name, timeout=None): - if name in dnscache: - return defer.succeed(dnscache[name]) - # in Twisted<=16.6, getHostByName() is always called with - # a default timeout of 60s (actually passed as (1, 3, 11, 45) tuple), - # so the input argument above is simply overridden - # to enforce Scrapy's DNS_TIMEOUT setting's value - timeout = (self.timeout,) - d = super(CachingThreadedResolver, self).getHostByName(name, timeout) - if dnscache.limit: - d.addCallback(self._cache_result, name) - return d + def resolveHostName(self, resolutionReceiver, hostName, portNumber=0, + addressTypes=None, transportSemantics='TCP'): - def _cache_result(self, result, name): - dnscache[name] = result - return result + @provider(IResolutionReceiver) + class CachingResolutionReceiver(resolutionReceiver): + def resolutionBegan(self, resolution): + super(CachingResolutionReceiver, self).resolutionBegan(resolution) + self.resolution = resolution + + def resolutionComplete(self): + super(CachingResolutionReceiver, self).resolutionComplete() + dnscache[hostName] = self.resolution + + try: + result = dnscache[hostName] + except KeyError: + result = self.resolver.resolveHostName( + CachingResolutionReceiver(), hostName, portNumber, addressTypes, transportSemantics + ) + finally: + return result From f1c184631e8cefc191dc0077083138c241bfd6a8 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 11 Dec 2019 17:44:05 -0300 Subject: [PATCH 136/313] Name resolver: timeout --- scrapy/resolver.py | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 03964f269..8792ed6ab 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -1,3 +1,4 @@ +from twisted.internet import reactor from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver from zope.interface.declarations import implementer, provider @@ -21,9 +22,14 @@ class CachingHostnameResolver(object): @provider(IResolutionReceiver) class CachingResolutionReceiver(resolutionReceiver): + + def __init__(self, timeout): + self.timeout = timeout + def resolutionBegan(self, resolution): super(CachingResolutionReceiver, self).resolutionBegan(resolution) self.resolution = resolution + # reactor.callLater(self.timeout, resolution.cancel) def resolutionComplete(self): super(CachingResolutionReceiver, self).resolutionComplete() @@ -33,7 +39,11 @@ class CachingHostnameResolver(object): result = dnscache[hostName] except KeyError: result = self.resolver.resolveHostName( - CachingResolutionReceiver(), hostName, portNumber, addressTypes, transportSemantics + CachingResolutionReceiver(self.timeout), + hostName, + portNumber, + addressTypes, + transportSemantics ) finally: return result From 55babf9acd6cd357f4211a86af01a1f2abe2f0cb Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 15 Jan 2020 12:25:20 -0300 Subject: [PATCH 137/313] Cache resolution only if the DNS request was successful --- scrapy/resolver.py | 10 ++++++++-- 1 file changed, 8 insertions(+), 2 deletions(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 8792ed6ab..0ba22ed0d 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -10,7 +10,7 @@ dnscache = LocalCache(10000) @implementer(IHostnameResolver) -class CachingHostnameResolver(object): +class CachingHostnameResolver: def __init__(self, resolver, cache_size, timeout): self.resolver = resolver @@ -25,15 +25,21 @@ class CachingHostnameResolver(object): def __init__(self, timeout): self.timeout = timeout + self.resolved = False def resolutionBegan(self, resolution): super(CachingResolutionReceiver, self).resolutionBegan(resolution) self.resolution = resolution # reactor.callLater(self.timeout, resolution.cancel) + def addressResolved(self, address): + super(CachingResolutionReceiver, self).addressResolved(address) + self.resolved = True + def resolutionComplete(self): super(CachingResolutionReceiver, self).resolutionComplete() - dnscache[hostName] = self.resolution + if self.resolved: + dnscache[hostName] = self.resolution try: result = dnscache[hostName] From 8c3de288fa2564473f49198b40efdeaa2428f1ca Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 15 Jan 2020 12:31:36 -0300 Subject: [PATCH 138/313] Remove non-working DNS timeout code --- scrapy/crawler.py | 1 - scrapy/resolver.py | 12 +++--------- 2 files changed, 3 insertions(+), 10 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 1350ea84f..61851acc3 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -320,7 +320,6 @@ class CrawlerProcess(CrawlerRunner): return CachingHostnameResolver( resolver=reactor.nameResolver, cache_size=cache_size, - timeout=self.settings.getfloat('DNS_TIMEOUT'), ) def _graceful_stop_reactor(self): diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 0ba22ed0d..ddbae61a9 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -1,4 +1,3 @@ -from twisted.internet import reactor from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver from zope.interface.declarations import implementer, provider @@ -12,9 +11,8 @@ dnscache = LocalCache(10000) @implementer(IHostnameResolver) class CachingHostnameResolver: - def __init__(self, resolver, cache_size, timeout): + def __init__(self, resolver, cache_size): self.resolver = resolver - self.timeout = timeout dnscache.limit = cache_size def resolveHostName(self, resolutionReceiver, hostName, portNumber=0, @@ -23,14 +21,10 @@ class CachingHostnameResolver: @provider(IResolutionReceiver) class CachingResolutionReceiver(resolutionReceiver): - def __init__(self, timeout): - self.timeout = timeout - self.resolved = False - def resolutionBegan(self, resolution): super(CachingResolutionReceiver, self).resolutionBegan(resolution) self.resolution = resolution - # reactor.callLater(self.timeout, resolution.cancel) + self.resolved = False def addressResolved(self, address): super(CachingResolutionReceiver, self).addressResolved(address) @@ -45,7 +39,7 @@ class CachingHostnameResolver: result = dnscache[hostName] except KeyError: result = self.resolver.resolveHostName( - CachingResolutionReceiver(self.timeout), + CachingResolutionReceiver(), hostName, portNumber, addressTypes, From e69cf415c8326349c871fb23faba2ed822aa08ee Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 03:58:07 -0300 Subject: [PATCH 139/313] Ability to choose name resolver --- scrapy/crawler.py | 14 ++------ scrapy/resolver.py | 54 ++++++++++++++++++++++++++++- scrapy/settings/default_settings.py | 1 + 3 files changed, 56 insertions(+), 13 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 61851acc3..c5351c08f 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -305,23 +305,13 @@ class CrawlerProcess(CrawlerRunner): return d.addBoth(self._stop_reactor) - reactor.installNameResolver(self._get_dns_resolver()) + resolver_class = load_object(self.settings["DNS_RESOLVER"]) + resolver_class.install(reactor, self.settings) tp = reactor.getThreadPool() tp.adjustPoolsize(maxthreads=self.settings.getint('REACTOR_THREADPOOL_MAXSIZE')) reactor.addSystemEventTrigger('before', 'shutdown', self.stop) reactor.run(installSignalHandlers=False) # blocking call - def _get_dns_resolver(self): - from twisted.internet import reactor - if self.settings.getbool('DNSCACHE_ENABLED'): - cache_size = self.settings.getint('DNSCACHE_SIZE') - else: - cache_size = 0 - return CachingHostnameResolver( - resolver=reactor.nameResolver, - cache_size=cache_size, - ) - def _graceful_stop_reactor(self): d = self.stop() d.addBoth(self._stop_reactor) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index ddbae61a9..2bef9f1b8 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -1,4 +1,6 @@ -from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver +from twisted.internet import defer +from twisted.internet.base import ThreadedResolver +from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver, IResolverSimple from zope.interface.declarations import implementer, provider from scrapy.utils.datatypes import LocalCache @@ -8,8 +10,58 @@ from scrapy.utils.datatypes import LocalCache dnscache = LocalCache(10000) +@implementer(IResolverSimple) +class CachingThreadedResolver(ThreadedResolver): + """ + Default caching resolver. IPv4 only, supports setting a timeout value for DNS requests + """ + + @classmethod + def install(cls, reactor, settings): + if settings.getbool('DNSCACHE_ENABLED'): + cache_size = settings.getint('DNSCACHE_SIZE') + else: + cache_size = 0 + resolver = cls(reactor, cache_size, settings.getfloat('DNS_TIMEOUT')) + reactor.installResolver(resolver) + + def __init__(self, reactor, cache_size, timeout): + super(CachingThreadedResolver, self).__init__(reactor) + dnscache.limit = cache_size + self.timeout = timeout + + def getHostByName(self, name, timeout=None): + if name in dnscache: + return defer.succeed(dnscache[name]) + # in Twisted<=16.6, getHostByName() is always called with + # a default timeout of 60s (actually passed as (1, 3, 11, 45) tuple), + # so the input argument above is simply overridden + # to enforce Scrapy's DNS_TIMEOUT setting's value + timeout = (self.timeout,) + d = super(CachingThreadedResolver, self).getHostByName(name, timeout) + if dnscache.limit: + d.addCallback(self._cache_result, name) + return d + + def _cache_result(self, result, name): + dnscache[name] = result + return result + + @implementer(IHostnameResolver) class CachingHostnameResolver: + """ + Experimental caching resolver, supporting IPv4 and IPv6 + """ + + @classmethod + def install(cls, reactor, settings): + if settings.getbool('DNSCACHE_ENABLED'): + cache_size = settings.getint('DNSCACHE_SIZE') + else: + cache_size = 0 + resolver = cls(reactor.nameResolver, cache_size) + reactor.installNameResolver(resolver) def __init__(self, resolver, cache_size): self.resolver = resolver diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index d03fd37b0..46ed3be96 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -60,6 +60,7 @@ DEPTH_PRIORITY = 0 DNSCACHE_ENABLED = True DNSCACHE_SIZE = 10000 +DNS_RESOLVER = 'scrapy.resolver.CachingThreadedResolver' DNS_TIMEOUT = 60 DOWNLOAD_DELAY = 0 From 0f155b059a43b2cc48149a26c6584910038a23f1 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 04:27:13 -0300 Subject: [PATCH 140/313] Make Flake8 happy (remove unused import) --- scrapy/crawler.py | 1 - 1 file changed, 1 deletion(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index c5351c08f..5658e264b 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -11,7 +11,6 @@ from scrapy.core.engine import ExecutionEngine from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.extension import ExtensionManager from scrapy.interfaces import ISpiderLoader -from scrapy.resolver import CachingHostnameResolver from scrapy.settings import overridden_settings, Settings from scrapy.signalmanager import SignalManager from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_reactor_installed From f45b4c7f8d191c6855c12f3497ece6b6121289df Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 10:09:34 -0300 Subject: [PATCH 141/313] from_crawler support for name resolvers --- scrapy/crawler.py | 2 +- scrapy/resolver.py | 48 ++++++++++++++++++++++++++++------------------ 2 files changed, 30 insertions(+), 20 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 5658e264b..4531c3aac 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -305,7 +305,7 @@ class CrawlerProcess(CrawlerRunner): d.addBoth(self._stop_reactor) resolver_class = load_object(self.settings["DNS_RESOLVER"]) - resolver_class.install(reactor, self.settings) + resolver_class.install_on_reactor(reactor, crawler=self) tp = reactor.getThreadPool() tp.adjustPoolsize(maxthreads=self.settings.getint('REACTOR_THREADPOOL_MAXSIZE')) reactor.addSystemEventTrigger('before', 'shutdown', self.stop) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 2bef9f1b8..792563e79 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -4,6 +4,7 @@ from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver, from zope.interface.declarations import implementer, provider from scrapy.utils.datatypes import LocalCache +from scrapy.utils.misc import create_instance # TODO: cache misses @@ -13,23 +14,27 @@ dnscache = LocalCache(10000) @implementer(IResolverSimple) class CachingThreadedResolver(ThreadedResolver): """ - Default caching resolver. IPv4 only, supports setting a timeout value for DNS requests + Default caching resolver. IPv4 only, supports setting a timeout value for DNS requests. """ - @classmethod - def install(cls, reactor, settings): - if settings.getbool('DNSCACHE_ENABLED'): - cache_size = settings.getint('DNSCACHE_SIZE') - else: - cache_size = 0 - resolver = cls(reactor, cache_size, settings.getfloat('DNS_TIMEOUT')) - reactor.installResolver(resolver) - def __init__(self, reactor, cache_size, timeout): super(CachingThreadedResolver, self).__init__(reactor) dnscache.limit = cache_size self.timeout = timeout + @classmethod + def from_crawler(cls, crawler, reactor): + if crawler.settings.getbool('DNSCACHE_ENABLED'): + cache_size = crawler.settings.getint('DNSCACHE_SIZE') + else: + cache_size = 0 + return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) + + @classmethod + def install_on_reactor(cls, reactor, crawler): + resolver = create_instance(cls, None, crawler, reactor=reactor) + reactor.installResolver(resolver) + def getHostByName(self, name, timeout=None): if name in dnscache: return defer.succeed(dnscache[name]) @@ -51,21 +56,26 @@ class CachingThreadedResolver(ThreadedResolver): @implementer(IHostnameResolver) class CachingHostnameResolver: """ - Experimental caching resolver, supporting IPv4 and IPv6 + Experimental caching resolver. Resolves IPv4 and IPv6 addresses, + does not support setting a timeout value for DNS requests. """ + def __init__(self, reactor, cache_size): + self.resolver = reactor.nameResolver + dnscache.limit = cache_size + @classmethod - def install(cls, reactor, settings): - if settings.getbool('DNSCACHE_ENABLED'): - cache_size = settings.getint('DNSCACHE_SIZE') + def from_crawler(cls, crawler, reactor): + if crawler.settings.getbool('DNSCACHE_ENABLED'): + cache_size = crawler.settings.getint('DNSCACHE_SIZE') else: cache_size = 0 - resolver = cls(reactor.nameResolver, cache_size) - reactor.installNameResolver(resolver) + return cls(reactor, cache_size) - def __init__(self, resolver, cache_size): - self.resolver = resolver - dnscache.limit = cache_size + @classmethod + def install_on_reactor(cls, reactor, crawler): + resolver = create_instance(cls, None, crawler, reactor=reactor) + reactor.installNameResolver(resolver) def resolveHostName(self, resolutionReceiver, hostName, portNumber=0, addressTypes=None, transportSemantics='TCP'): From 3cfa73b8b12dfe2dd1364af856d2ec46a486b48d Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 18:01:18 -0300 Subject: [PATCH 142/313] Name resolvers: install_on_reactor as instance method --- scrapy/crawler.py | 5 +++-- scrapy/resolver.py | 13 ++++--------- 2 files changed, 7 insertions(+), 11 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 4531c3aac..0cef75a6a 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -21,7 +21,7 @@ from scrapy.utils.log import ( log_scrapy_info, LogCounterHandler, ) -from scrapy.utils.misc import load_object +from scrapy.utils.misc import create_instance, load_object from scrapy.utils.ossignal import install_shutdown_handlers, signal_names @@ -305,7 +305,8 @@ class CrawlerProcess(CrawlerRunner): d.addBoth(self._stop_reactor) resolver_class = load_object(self.settings["DNS_RESOLVER"]) - resolver_class.install_on_reactor(reactor, crawler=self) + resolver = create_instance(resolver_class, self.settings, self, reactor=reactor) + resolver.install_on_reactor(reactor) tp = reactor.getThreadPool() tp.adjustPoolsize(maxthreads=self.settings.getint('REACTOR_THREADPOOL_MAXSIZE')) reactor.addSystemEventTrigger('before', 'shutdown', self.stop) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 792563e79..2b97603da 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -4,7 +4,6 @@ from twisted.internet.interfaces import IHostnameResolver, IResolutionReceiver, from zope.interface.declarations import implementer, provider from scrapy.utils.datatypes import LocalCache -from scrapy.utils.misc import create_instance # TODO: cache misses @@ -30,10 +29,8 @@ class CachingThreadedResolver(ThreadedResolver): cache_size = 0 return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) - @classmethod - def install_on_reactor(cls, reactor, crawler): - resolver = create_instance(cls, None, crawler, reactor=reactor) - reactor.installResolver(resolver) + def install_on_reactor(self, reactor): + reactor.installResolver(self) def getHostByName(self, name, timeout=None): if name in dnscache: @@ -72,10 +69,8 @@ class CachingHostnameResolver: cache_size = 0 return cls(reactor, cache_size) - @classmethod - def install_on_reactor(cls, reactor, crawler): - resolver = create_instance(cls, None, crawler, reactor=reactor) - reactor.installNameResolver(resolver) + def install_on_reactor(self, reactor): + reactor.installNameResolver(self) def resolveHostName(self, resolutionReceiver, hostName, portNumber=0, addressTypes=None, transportSemantics='TCP'): From 1040f581ec1ba3bcf8f08edb824d59c0e4a93700 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 20:14:52 -0300 Subject: [PATCH 143/313] Name resolvers: do not pass the reactor to the install method --- scrapy/crawler.py | 2 +- scrapy/resolver.py | 14 ++++++++------ 2 files changed, 9 insertions(+), 7 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 0cef75a6a..35c6b7716 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -306,7 +306,7 @@ class CrawlerProcess(CrawlerRunner): resolver_class = load_object(self.settings["DNS_RESOLVER"]) resolver = create_instance(resolver_class, self.settings, self, reactor=reactor) - resolver.install_on_reactor(reactor) + resolver.install_on_reactor() tp = reactor.getThreadPool() tp.adjustPoolsize(maxthreads=self.settings.getint('REACTOR_THREADPOOL_MAXSIZE')) reactor.addSystemEventTrigger('before', 'shutdown', self.stop) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 2b97603da..7c776f75e 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -19,6 +19,7 @@ class CachingThreadedResolver(ThreadedResolver): def __init__(self, reactor, cache_size, timeout): super(CachingThreadedResolver, self).__init__(reactor) dnscache.limit = cache_size + self.reactor = reactor self.timeout = timeout @classmethod @@ -29,8 +30,8 @@ class CachingThreadedResolver(ThreadedResolver): cache_size = 0 return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) - def install_on_reactor(self, reactor): - reactor.installResolver(self) + def install_on_reactor(self,): + self.reactor.installResolver(self) def getHostByName(self, name, timeout=None): if name in dnscache: @@ -58,7 +59,8 @@ class CachingHostnameResolver: """ def __init__(self, reactor, cache_size): - self.resolver = reactor.nameResolver + self.reactor = reactor + self.original_resolver = reactor.nameResolver dnscache.limit = cache_size @classmethod @@ -69,8 +71,8 @@ class CachingHostnameResolver: cache_size = 0 return cls(reactor, cache_size) - def install_on_reactor(self, reactor): - reactor.installNameResolver(self) + def install_on_reactor(self): + self.reactor.installNameResolver(self) def resolveHostName(self, resolutionReceiver, hostName, portNumber=0, addressTypes=None, transportSemantics='TCP'): @@ -95,7 +97,7 @@ class CachingHostnameResolver: try: result = dnscache[hostName] except KeyError: - result = self.resolver.resolveHostName( + result = self.original_resolver.resolveHostName( CachingResolutionReceiver(), hostName, portNumber, From 90e3bd8715701aeea9187837bc32925e3225288f Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 20:32:40 -0300 Subject: [PATCH 144/313] [test] Name resolvers --- tests/CrawlerProcess/alternative_name_resolver.py | 15 +++++++++++++++ tests/CrawlerProcess/default_name_resolver.py | 12 ++++++++++++ tests/test_crawler.py | 12 ++++++++++++ 3 files changed, 39 insertions(+) create mode 100644 tests/CrawlerProcess/alternative_name_resolver.py create mode 100644 tests/CrawlerProcess/default_name_resolver.py diff --git a/tests/CrawlerProcess/alternative_name_resolver.py b/tests/CrawlerProcess/alternative_name_resolver.py new file mode 100644 index 000000000..2c466da04 --- /dev/null +++ b/tests/CrawlerProcess/alternative_name_resolver.py @@ -0,0 +1,15 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class IPv6Spider(scrapy.Spider): + name = "ipv6_spider" + start_urls = ["http://[::1]"] + + +process = CrawlerProcess(settings={ + "RETRY_ENABLED": False, + "DNS_RESOLVER": "scrapy.resolver.CachingHostnameResolver", +}) +process.crawl(IPv6Spider) +process.start() diff --git a/tests/CrawlerProcess/default_name_resolver.py b/tests/CrawlerProcess/default_name_resolver.py new file mode 100644 index 000000000..60d91b68b --- /dev/null +++ b/tests/CrawlerProcess/default_name_resolver.py @@ -0,0 +1,12 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class IPv6Spider(scrapy.Spider): + name = "ipv6_spider" + start_urls = ["http://[::1]"] + + +process = CrawlerProcess(settings={"RETRY_ENABLED": False}) +process.crawl(IPv6Spider) +process.start() diff --git a/tests/test_crawler.py b/tests/test_crawler.py index fce60ca37..d85f2bc41 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -305,3 +305,15 @@ class CrawlerProcessSubprocess(unittest.TestCase): log = self.run_script('asyncio_enabled_reactor.py') self.assertIn('Spider closed (finished)', log) self.assertIn("DEBUG: Asyncio reactor is installed", log) + + def test_default_name_resolver(self): + log = self.run_script('default_name_resolver.py') + self.assertIn('Spider closed (finished)', log) + self.assertIn("twisted.internet.error.DNSLookupError: DNS lookup failed: no results for hostname lookup: ::1.", log) + self.assertIn("'downloader/exception_type_count/twisted.internet.error.DNSLookupError': 1,", log) + + def test_alternative_name_resolver(self): + log = self.run_script('alternative_name_resolver.py') + self.assertIn('Spider closed (finished)', log) + self.assertIn("twisted.internet.error.ConnectionRefusedError: Connection was refused by other side: 111: Connection refused.", log) + self.assertIn("'downloader/exception_type_count/twisted.internet.error.ConnectionRefusedError': 1,", log) From d487498cff893d2c507ef5ed0957f101dc3ab4a9 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 22:02:01 -0300 Subject: [PATCH 145/313] Update name resolvers tests --- tests/test_crawler.py | 14 ++++++++++---- 1 file changed, 10 insertions(+), 4 deletions(-) diff --git a/tests/test_crawler.py b/tests/test_crawler.py index d85f2bc41..14f4f8418 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -306,14 +306,20 @@ class CrawlerProcessSubprocess(unittest.TestCase): self.assertIn('Spider closed (finished)', log) self.assertIn("DEBUG: Asyncio reactor is installed", log) - def test_default_name_resolver(self): + def test_ipv6_default_name_resolver(self): log = self.run_script('default_name_resolver.py') self.assertIn('Spider closed (finished)', log) self.assertIn("twisted.internet.error.DNSLookupError: DNS lookup failed: no results for hostname lookup: ::1.", log) self.assertIn("'downloader/exception_type_count/twisted.internet.error.DNSLookupError': 1,", log) - def test_alternative_name_resolver(self): + def test_ipv6_alternative_name_resolver(self): log = self.run_script('alternative_name_resolver.py') self.assertIn('Spider closed (finished)', log) - self.assertIn("twisted.internet.error.ConnectionRefusedError: Connection was refused by other side: 111: Connection refused.", log) - self.assertIn("'downloader/exception_type_count/twisted.internet.error.ConnectionRefusedError': 1,", log) + self.assertTrue(any( + "twisted.internet.error.ConnectionRefusedError" in log, + "twisted.internet.error.ConnectError" in log, + )) + self.assertTrue(any( + "'downloader/exception_type_count/twisted.internet.error.ConnectionRefusedError': 1," in log, + "'downloader/exception_type_count/twisted.internet.error.ConnectError': 1," in log, + )) From dee420a69cb5c8319a7df99a5765c5ca90a7063e Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 16 Jan 2020 23:48:16 -0300 Subject: [PATCH 146/313] Fix name resolvers tests --- tests/test_crawler.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 14f4f8418..0ce0674de 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -315,11 +315,11 @@ class CrawlerProcessSubprocess(unittest.TestCase): def test_ipv6_alternative_name_resolver(self): log = self.run_script('alternative_name_resolver.py') self.assertIn('Spider closed (finished)', log) - self.assertTrue(any( + self.assertTrue(any([ "twisted.internet.error.ConnectionRefusedError" in log, "twisted.internet.error.ConnectError" in log, - )) - self.assertTrue(any( + ])) + self.assertTrue(any([ "'downloader/exception_type_count/twisted.internet.error.ConnectionRefusedError': 1," in log, "'downloader/exception_type_count/twisted.internet.error.ConnectError': 1," in log, - )) + ])) From 41f7ebf3add25b2251082b206b062c00d1daba58 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 17 Jan 2020 12:40:49 -0300 Subject: [PATCH 147/313] CachingThreadedResolver: No need to store the reactor as an instance attribute It's already done in the parent class --- scrapy/resolver.py | 1 - 1 file changed, 1 deletion(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 7c776f75e..7751f3796 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -19,7 +19,6 @@ class CachingThreadedResolver(ThreadedResolver): def __init__(self, reactor, cache_size, timeout): super(CachingThreadedResolver, self).__init__(reactor) dnscache.limit = cache_size - self.reactor = reactor self.timeout = timeout @classmethod From 302d3f552b5a930a43fddac8eb8c5f769d8748bc Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sat, 18 Jan 2020 01:41:57 -0300 Subject: [PATCH 148/313] [doc] DNS_RESOLVER setting --- docs/topics/settings.rst | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index c02f877fc..292eaea74 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -397,6 +397,19 @@ Default: ``10000`` DNS in-memory cache size. +.. setting:: DNS_RESOLVER + +DNS_RESOLVER +------------ + +Default: ``'scrapy.resolver.CachingThreadedResolver'`` + +The class to be used to resolve DNS names. The default ``scrapy.resolver.CachingThreadedResolver`` +supports specifying a timeout for DNS requests via the :setting:`DNS_TIMEOUT` setting, +but works only with IPv4 addresses. Scrapy provides an alternative resolver, +``scrapy.resolver.CachingHostnameResolver``, which supports IPv4/IPv6 addresses but does not +take the :setting:`DNS_TIMEOUT` setting into account. + .. setting:: DNS_TIMEOUT DNS_TIMEOUT From b471765d40a2a8a94f90af46ac6903e3786aec5f Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sat, 18 Jan 2020 01:52:29 -0300 Subject: [PATCH 149/313] [doc] FAQ entry about the IPv6 and the DNS_RESOLVER setting --- docs/faq.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/faq.rst b/docs/faq.rst index aae2411e0..b789a8cdb 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -353,6 +353,13 @@ method for this purpose. For example:: for _ in range(item['multiply_by']): yield deepcopy(item) +Does Scrapy support IPv6 addresses? +----------------------------------- + +Yes, by setting :setting:`DNS_RESOLVER` to ``scrapy.resolver.CachingHostnameResolver``. +Note that by doing so, you lose the ability to set a specific timeout for DNS requests +(the value of the :setting:`DNS_TIMEOUT` setting is ignored). + .. _user agents: https://en.wikipedia.org/wiki/User_agent .. _LIFO: https://en.wikipedia.org/wiki/Stack_(abstract_data_type) From eaa8ed02d05d0cc5e7be1f773a22755676c7fe56 Mon Sep 17 00:00:00 2001 From: Juan Pablo Balarini <jpbalarini@gmail.com> Date: Wed, 26 Dec 2018 13:11:23 -0300 Subject: [PATCH 150/313] Add ability to change max_active_size by settings --- docs/topics/settings.rst | 11 +++++++++++ scrapy/core/scraper.py | 2 +- scrapy/settings/default_settings.py | 2 ++ 3 files changed, 14 insertions(+), 1 deletion(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index c02f877fc..d4480c879 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1262,6 +1262,17 @@ Type of priority queue used by the scheduler. Another available type is domains in parallel. But currently ``scrapy.pqueues.DownloaderAwarePriorityQueue`` does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`. +.. setting:: SCRAPER_SLOT_MAX_ACTIVE_SIZE + +SCRAPER_SLOT_MAX_ACTIVE_SIZE +---------------------------- +Default: ``5000000`` + +Soft limit (in bytes) for response data being processed. + +While the sum of the sizes of all responses being processed is above this value, +Scrapy does not process new requests. + .. setting:: SPIDER_CONTRACTS SPIDER_CONTRACTS diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 99114d3bb..facbd8b73 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -78,7 +78,7 @@ class Scraper(object): @defer.inlineCallbacks def open_spider(self, spider): """Open the given spider for scraping and allocate resources for it""" - self.slot = Slot() + self.slot = Slot(self.crawler.settings.getint('SCRAPER_SLOT_MAX_ACTIVE_SIZE')) yield self.itemproc.open_spider(spider) def close_spider(self, spider): diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index d03fd37b0..ddd48d327 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -254,6 +254,8 @@ SCHEDULER_DISK_QUEUE = 'scrapy.squeues.PickleLifoDiskQueue' SCHEDULER_MEMORY_QUEUE = 'scrapy.squeues.LifoMemoryQueue' SCHEDULER_PRIORITY_QUEUE = 'scrapy.pqueues.ScrapyPriorityQueue' +SCRAPER_SLOT_MAX_ACTIVE_SIZE = 5000000 + SPIDER_LOADER_CLASS = 'scrapy.spiderloader.SpiderLoader' SPIDER_LOADER_WARN_ONLY = False From 0f2d871d88b6741aaf0348d1f11a1e020124099f Mon Sep 17 00:00:00 2001 From: JP Balarini <jpbalarini@gmail.com> Date: Wed, 22 May 2019 16:26:27 -0300 Subject: [PATCH 151/313] Use PEP 515 style for SCRAPER_SLOT_MAX_ACTIVE_SIZE documentation --- docs/topics/settings.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index d4480c879..f4c3494f5 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1266,7 +1266,7 @@ does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`. SCRAPER_SLOT_MAX_ACTIVE_SIZE ---------------------------- -Default: ``5000000`` +Default: ``5_000_000`` Soft limit (in bytes) for response data being processed. From 8ea8f14827470f37c0e53d302aa65bcfa9604f3c Mon Sep 17 00:00:00 2001 From: OmarFarrag <omar.alaa.farrag@gmail.com> Date: Mon, 20 Jan 2020 18:19:36 +0200 Subject: [PATCH 152/313] Update scrapy/utils/ftp.py Co-Authored-By: Mikhail Korobov <kmike84@gmail.com> --- scrapy/utils/ftp.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index bf67b9976..b3e9ec2ed 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -17,7 +17,7 @@ def ftp_makedirs_cwd(ftp, path, first_call=True): ftp.cwd(path) def ftp_store_file( - path, file, host ,port, + path, file, host, port, username, password, use_active_mode=False): """Opens a FTP connection with passed credentials,sets current directory to the directory extracted from given path, then uploads the file to server From 06ab668ec7f880f9992dc669a374a6111cef5d04 Mon Sep 17 00:00:00 2001 From: OmarFarrag <omar.alaa.farrag@gmail.com> Date: Wed, 22 Jan 2020 03:48:07 +0200 Subject: [PATCH 153/313] Use kwargs-only parameters in `ftp_store_file` --- scrapy/extensions/feedexport.py | 6 +++--- scrapy/pipelines/files.py | 6 +++--- scrapy/utils/ftp.py | 2 +- 3 files changed, 7 insertions(+), 7 deletions(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 1ddc55f93..06b5a0dd9 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -175,9 +175,9 @@ class FTPFeedStorage(BlockingFeedStorage): def _store_in_thread(self, file): ftp_store_file( - self.path, file, self.host, - self.port, self.username, - self.password, self.use_active_mode + path=self.path, file=file, host=self.host, + port=self.port, username=self.username, + password=self.password, use_active_mode=self.use_active_mode ) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 5780f63bd..5383b05fe 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -269,9 +269,9 @@ class FTPFilesStore(object): def persist_file(self, path, buf, info, meta=None, headers=None): path = '%s/%s' % (self.basedir, path) return threads.deferToThread( - ftp_store_file, path,buf, - self.host, self.port,self.username, - self.password, self.USE_ACTIVE_MODE + ftp_store_file, path=path, file=buf, + host=self.host, port=self.port, username=self.username, + password=self.password, use_active_mode=self.USE_ACTIVE_MODE ) def stat_file(self, path, info): diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index b3e9ec2ed..752e3c953 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -17,7 +17,7 @@ def ftp_makedirs_cwd(ftp, path, first_call=True): ftp.cwd(path) def ftp_store_file( - path, file, host, port, + *, path, file, host, port, username, password, use_active_mode=False): """Opens a FTP connection with passed credentials,sets current directory to the directory extracted from given path, then uploads the file to server From c75cf15b7a8293fd83f6e8b27abca2e50b0a04ed Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 22 Jan 2020 10:38:59 -0300 Subject: [PATCH 154/313] Update CSS selectors in tutorial --- docs/intro/tutorial.rst | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index c4d8d717b..c9d00eb74 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -616,19 +616,24 @@ instance; you still have to yield this Request. You can also pass a selector to ``response.follow`` instead of a string; this selector should extract necessary attributes:: - href = response.css('li.next a::attr(href)')[0] - yield response.follow(href, callback=self.parse) + for href in response.css('ul.pager a::attr(href)'): + yield response.follow(href, callback=self.parse) For ``<a>`` elements there is a shortcut: ``response.follow`` uses their href attribute automatically. So the code can be shortened further:: - a = response.css('li.next a')[0] - yield response.follow(a, callback=self.parse) + for a in response.css('ul.pager a'): + yield response.follow(a, callback=self.parse) To create multiple requests from an iterable, you can use :meth:`response.follow_all <scrapy.http.TextResponse.follow_all>` instead:: - yield from response.follow_all(response.css('a'), callback=self.parse) + anchors = response.css('ul.pager a') + yield from response.follow_all(anchors, callback=self.parse) + +or, shortening it further:: + + yield from response.follow_all(css='ul.pager a', callback=self.parse) More examples and patterns From 7d5cebcf773d54d2f7b5ac0b90886847a4a70c26 Mon Sep 17 00:00:00 2001 From: Peter Vandenabeele <peter@vandenabeele.com> Date: Thu, 23 Jan 2020 09:08:21 +0100 Subject: [PATCH 155/313] fix logical documentation error with PER_DOMAIN or PER_DOMAIN --- docs/faq.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/faq.rst b/docs/faq.rst index aae2411e0..169e1a479 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -140,7 +140,7 @@ setting the following settings:: While pending requests are below the configured values of :setting:`CONCURRENT_REQUESTS`, :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` or -:setting:`CONCURRENT_REQUESTS_PER_DOMAIN`, those requests are sent +:setting:`CONCURRENT_REQUESTS_PER_IP`, those requests are sent concurrently. As a result, the first few requests of a crawl rarely follow the desired order. Lowering those settings to ``1`` enforces the desired order, but it significantly slows down the crawl as a whole. From 9899414300b4a6491ea17418c3403aed08e0faf6 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 23 Jan 2020 18:06:59 -0300 Subject: [PATCH 156/313] Name resolver: return result directly --- scrapy/resolver.py | 6 ++---- 1 file changed, 2 insertions(+), 4 deletions(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 7751f3796..554a3a14d 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -94,14 +94,12 @@ class CachingHostnameResolver: dnscache[hostName] = self.resolution try: - result = dnscache[hostName] + return dnscache[hostName] except KeyError: - result = self.original_resolver.resolveHostName( + return self.original_resolver.resolveHostName( CachingResolutionReceiver(), hostName, portNumber, addressTypes, transportSemantics ) - finally: - return result From c544c0d2b8356125d1a5465b44617aaaaeab0ea1 Mon Sep 17 00:00:00 2001 From: OmarFarrag <omar.alaa.farrag@gmail.com> Date: Fri, 24 Jan 2020 14:36:16 +0200 Subject: [PATCH 157/313] Use context management with `FTP` --- scrapy/utils/ftp.py | 19 +++++++++---------- 1 file changed, 9 insertions(+), 10 deletions(-) diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index 752e3c953..9992a916e 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -22,13 +22,12 @@ def ftp_store_file( """Opens a FTP connection with passed credentials,sets current directory to the directory extracted from given path, then uploads the file to server """ - ftp = FTP() - ftp.connect(host, port) - ftp.login(username, password) - if use_active_mode: - ftp.set_pasv(False) - file.seek(0) - dirname, filename = posixpath.split(path) - ftp_makedirs_cwd(ftp, dirname) - ftp.storbinary('STOR %s' % filename, file) - ftp.quit() + with FTP() as ftp: + ftp.connect(host, port) + ftp.login(username, password) + if use_active_mode: + ftp.set_pasv(False) + file.seek(0) + dirname, filename = posixpath.split(path) + ftp_makedirs_cwd(ftp, dirname) + ftp.storbinary('STOR %s' % filename, file) From f5d9eb15f8b50c64c44a7f859a953b92d7a33e6b Mon Sep 17 00:00:00 2001 From: OmarFarrag <omar.alaa.farrag@gmail.com> Date: Fri, 24 Jan 2020 15:06:40 +0200 Subject: [PATCH 158/313] use `__future__` imports at the begining of the file --- scrapy/utils/test.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/utils/test.py b/scrapy/utils/test.py index fd8f41137..65d24314e 100644 --- a/scrapy/utils/test.py +++ b/scrapy/utils/test.py @@ -2,10 +2,10 @@ This module contains some assorted functions used in tests """ -import asyncio -import os from __future__ import absolute_import from posixpath import split +import asyncio +import os from importlib import import_module from twisted.trial.unittest import SkipTest From 40e0a11aa8dd499f725aaa206643aa36411fd514 Mon Sep 17 00:00:00 2001 From: OmarFarrag <omar.alaa.farrag@gmail.com> Date: Fri, 24 Jan 2020 15:51:48 +0200 Subject: [PATCH 159/313] Fix Flake8 errors --- scrapy/extensions/feedexport.py | 6 ++---- scrapy/pipelines/files.py | 18 +++++++++--------- scrapy/utils/ftp.py | 3 ++- scrapy/utils/test.py | 8 +++++--- tests/test_pipeline_files.py | 5 ++++- 5 files changed, 22 insertions(+), 18 deletions(-) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index c1ca0ca67..f1b101780 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -7,18 +7,16 @@ See documentation in docs/topics/feed-exports.rst import os import sys import logging -import posixpath from tempfile import NamedTemporaryFile from datetime import datetime from urllib.parse import urlparse, unquote -from ftplib import FTP from zope.interface import Interface, implementer from twisted.internet import defer, threads from w3lib.url import file_uri_to_path from scrapy import signals -from scrapy.utils.ftp import ftp_makedirs_cwd, ftp_store_file +from scrapy.utils.ftp import ftp_store_file from scrapy.exceptions import NotConfigured from scrapy.utils.misc import create_instance, load_object from scrapy.utils.log import failure_to_exc_info @@ -175,7 +173,7 @@ class FTPFeedStorage(BlockingFeedStorage): def _store_in_thread(self, file): ftp_store_file( path=self.path, file=file, host=self.host, - port=self.port, username=self.username, + port=self.port, username=self.username, password=self.password, use_active_mode=self.use_active_mode ) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 2d286eed8..7e9b12c0e 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -28,7 +28,7 @@ from scrapy.utils.python import to_bytes from scrapy.utils.request import referer_str from scrapy.utils.boto import is_botocore from scrapy.utils.datatypes import CaselessDict -from scrapy.utils.ftp import ftp_makedirs_cwd, ftp_store_file +from scrapy.utils.ftp import ftp_store_file logger = logging.getLogger(__name__) @@ -265,25 +265,25 @@ class FTPFilesStore(object): FTP_USERNAME = None FTP_PASSWORD = None USE_ACTIVE_MODE = None - + def __init__(self, uri): assert uri.startswith('ftp://') - u = urlparse(uri) + u = urlparse(uri) self.port = u.port self.host = u.hostname self.port = int(u.port or 21) self.username = u.username or self.FTP_USERNAME self.password = u.password or self.FTP_PASSWORD self.basedir = u.path.rstrip('/') - - def persist_file(self, path, buf, info, meta=None, headers=None): + + def persist_file(self, path, buf, info, meta=None, headers=None): path = '%s/%s' % (self.basedir, path) return threads.deferToThread( ftp_store_file, path=path, file=buf, host=self.host, port=self.port, username=self.username, password=self.password, use_active_mode=self.USE_ACTIVE_MODE ) - + def stat_file(self, path, info): def _stat_file(path): try: @@ -298,8 +298,8 @@ class FTPFilesStore(object): ftp.retrbinary('RETR %s' % file_path, m.update) return {'last_modified': last_modified, 'checksum': m.hexdigest()} # The file doesn't exist - except Exception as e : - return {} + except Exception: + return {} return threads.deferToThread(_stat_file, path) @@ -381,7 +381,7 @@ class FilesPipeline(MediaPipeline): ftp_store.FTP_USERNAME = settings['FTP_USER'] ftp_store.FTP_PASSWORD = settings['FTP_PASSWORD'] ftp_store.USE_ACTIVE_MODE = settings.getbool('FEED_STORAGE_FTP_ACTIVE') - + store_uri = settings['FILES_STORE'] return cls(store_uri, settings=settings) diff --git a/scrapy/utils/ftp.py b/scrapy/utils/ftp.py index 1bb754a69..f07bdd748 100644 --- a/scrapy/utils/ftp.py +++ b/scrapy/utils/ftp.py @@ -17,7 +17,8 @@ def ftp_makedirs_cwd(ftp, path, first_call=True): if first_call: ftp.cwd(path) -def ftp_store_file( + +def ftp_store_file( *, path, file, host, port, username, password, use_active_mode=False): """Opens a FTP connection with passed credentials,sets current directory diff --git a/scrapy/utils/test.py b/scrapy/utils/test.py index 65d24314e..61f2d059d 100644 --- a/scrapy/utils/test.py +++ b/scrapy/utils/test.py @@ -66,8 +66,9 @@ def get_gcs_content_and_delete(bucket, path): return content, acl, blob -def get_ftp_content_and_delete(path, host ,port, - username, password, use_active_mode=False): +def get_ftp_content_and_delete( + path, host, port,username, + password, use_active_mode=False): from ftplib import FTP ftp = FTP() ftp.connect(host, port) @@ -75,6 +76,7 @@ def get_ftp_content_and_delete(path, host ,port, if use_active_mode: ftp.set_pasv(False) ftp_data = [] + def buffer_data(data): ftp_data.append(data) ftp.retrbinary('RETR %s' % path, buffer_data) @@ -82,7 +84,7 @@ def get_ftp_content_and_delete(path, host ,port, ftp.cwd(dirname) ftp.delete(filename) return "".join(ftp_data) - + def get_crawler(spidercls=None, settings_dict=None): """Return an unconfigured Crawler object. If settings_dict is given, it diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index fc0453a97..e5bad2ed0 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -367,6 +367,7 @@ class TestGCSFilesStore(unittest.TestCase): self.assertEqual(blob.content_type, 'application/octet-stream') self.assertIn(expected_policy, acl) + class TestFTPFileStore(unittest.TestCase): @defer.inlineCallbacks def test_persist(self): @@ -386,10 +387,12 @@ class TestFTPFileStore(unittest.TestCase): self.assertIn('checksum', stat) self.assertEqual(stat['checksum'], 'd113d66b2ec7258724a268bd88eef6b6') path = '%s/%s' % (store.basedir, path) - content = get_ftp_content_and_delete(path, store.host, store.port, + content = get_ftp_content_and_delete( + path, store.host, store.port, store.username, store.password, store.USE_ACTIVE_MODE) self.assertEqual(data.decode(), content) + class ItemWithFiles(Item): file_urls = Field() files = Field() From 9e6d5573f1180bc70d7eca9f381204c238b3a550 Mon Sep 17 00:00:00 2001 From: OmarFarrag <omar.alaa.farrag@gmail.com> Date: Fri, 24 Jan 2020 15:58:52 +0200 Subject: [PATCH 160/313] Fix Flake8 errors --- scrapy/pipelines/files.py | 3 +-- scrapy/utils/test.py | 2 +- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 7e9b12c0e..9b7445755 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -13,7 +13,6 @@ from collections import defaultdict from email.utils import parsedate_tz, mktime_tz from ftplib import FTP from io import BytesIO -from six.moves.urllib.parse import urlparse from urllib.parse import urlparse from twisted.internet import defer, threads @@ -276,7 +275,7 @@ class FTPFilesStore(object): self.password = u.password or self.FTP_PASSWORD self.basedir = u.path.rstrip('/') - def persist_file(self, path, buf, info, meta=None, headers=None): + def persist_file(self, path, buf, info, meta=None, headers=None): path = '%s/%s' % (self.basedir, path) return threads.deferToThread( ftp_store_file, path=path, file=buf, diff --git a/scrapy/utils/test.py b/scrapy/utils/test.py index 61f2d059d..faac0b12f 100644 --- a/scrapy/utils/test.py +++ b/scrapy/utils/test.py @@ -67,7 +67,7 @@ def get_gcs_content_and_delete(bucket, path): def get_ftp_content_and_delete( - path, host, port,username, + path, host, port, username, password, use_active_mode=False): from ftplib import FTP ftp = FTP() From f3374a50479fcfd9395919b204bcaff2405f7148 Mon Sep 17 00:00:00 2001 From: Peter Vandenabeele <peter@vandenabeele.com> Date: Sat, 25 Jan 2020 16:52:30 +0100 Subject: [PATCH 161/313] Fix variable name `author_page_links` I did not test this code, but the change from `href` to this author_page_links seems to have a typo ? --- docs/intro/tutorial.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index c9d00eb74..ee10048b5 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -652,7 +652,7 @@ this time for scraping author information:: def parse(self, response): author_page_links = response.css('.author + a') - yield from response.follow_all(author_links, self.parse_author) + yield from response.follow_all(author_page_links, self.parse_author) pagination_links = response.css('li.next a') yield from response.follow_all(pagination_links, self.parse) From f72d4e93e6b8a6774c515d55ca73fdbd6a01f13b Mon Sep 17 00:00:00 2001 From: Peter Vandenabeele <peter@vandenabeele.com> Date: Sun, 26 Jan 2020 10:48:28 +0100 Subject: [PATCH 162/313] [Docs] 2 typos + 1 clarification in docs Fixing 2 small typos and adding 1 word as clarification in the downloader-middlewares. Also, I was confused with the entries like `ref:Reppy <reppy-parser>` and similar entries. Are these supposed to be links to other parts of the doc, or is this the intended way of showing these references ? --- docs/topics/downloader-middleware.rst | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index ae6d41809..8a760e53b 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -199,7 +199,7 @@ CookiesMiddleware This middleware enables working with sites that require cookies, such as those that use sessions. It keeps track of cookies sent by web servers, and - send them back on subsequent requests (from that spider), just like web + sends them back on subsequent requests (from that spider), just like web browsers do. The following settings can be used to configure the cookie middleware: @@ -672,7 +672,7 @@ sometimes a more nuanced policy is desirable. This setting still respects ``Cache-Control: no-store`` directives in responses. If you don't want that, filter ``no-store`` out of the Cache-Control headers in -responses you feedto the cache middleware. +responses you feed to the cache middleware. .. setting:: HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS @@ -686,7 +686,7 @@ Default: ``[]`` List of Cache-Control directives in responses to be ignored. Sites often set "no-store", "no-cache", "must-revalidate", etc., but get -upset at the traffic a spider can generate if it respects those +upset at the traffic a spider can generate if it actually respects those directives. This allows to selectively ignore Cache-Control directives that are known to be unimportant for the sites being crawled. From 80925ab845b7f55be97d9bb91015ceee90efc333 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 5 Aug 2019 11:39:07 -0300 Subject: [PATCH 163/313] Get server IP address for HTTP/1.1 responses --- docs/topics/request-response.rst | 12 +++++++++- scrapy/core/downloader/__init__.py | 2 +- scrapy/core/downloader/handlers/http11.py | 18 ++++++++++----- scrapy/http/response/__init__.py | 5 +++-- tests/test_crawl.py | 27 +++++++++++++++++++++++ 5 files changed, 54 insertions(+), 10 deletions(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 8997a7f19..a4cc1a7d7 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -34,7 +34,7 @@ Request objects :type url: string :param callback: the function that will be called with the response of this - request (once its downloaded) as its first parameter. For more information + request (once it's downloaded) as its first parameter. For more information see :ref:`topics-request-response-ref-request-callback-arguments` below. If a Request doesn't specify a callback, the spider's :meth:`~scrapy.spiders.Spider.parse` method will be used. @@ -611,6 +611,12 @@ Response objects This represents the :class:`Request` that generated this response. :type request: :class:`Request` object + :param ip_address: The IP address of the server from which the Response originated. + :type ip_address: :class:`ipaddress.IPv4Address` object + + .. FIXME: Add ipaddress.IPv6Address once it's supported + + .. attribute:: Response.url A string containing the URL of the response. @@ -679,6 +685,10 @@ Response objects they're shown on the string representation of the Response (`__str__` method) which is used by the engine for logging. + .. attribute:: Response.ip_address + + The IP address of the server from which the Response originated. + .. method:: Response.copy() Returns a new Response which is a copy of this Response. diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index 157dc3418..11c9dd908 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -172,7 +172,7 @@ class Downloader(object): return response dfd.addCallback(_downloaded) - # 3. After response arrives, remove the request from transferring + # 3. After response arrives, remove the request from transferring # state to free up the transferring slot so it can be used by the # following requests (perhaps those which came from the downloader # middleware itself) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 5a5f6cf0a..b690f439f 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -4,6 +4,7 @@ import logging import re import warnings from io import BytesIO +from ipaddress import ip_address from time import time from urllib.parse import urldefrag @@ -382,7 +383,7 @@ class ScrapyAgent(object): def _cb_bodyready(self, txresponse, request): # deliverBody hangs for responses without body if txresponse.length == 0: - return txresponse, b'', None + return txresponse, b'', None, None maxsize = request.meta.get('download_maxsize', self._maxsize) warnsize = request.meta.get('download_warnsize', self._warnsize) @@ -418,11 +419,11 @@ class ScrapyAgent(object): return d def _cb_bodydone(self, result, request, url): - txresponse, body, flags = result + txresponse, body, flags, ip_address = result status = int(txresponse.code) headers = Headers(txresponse.headers.getAllRawHeaders()) respcls = responsetypes.from_args(headers=headers, url=url, body=body) - return respcls(url=url, status=status, headers=headers, body=body, flags=flags) + return respcls(url=url, status=status, headers=headers, body=body, flags=flags, ip_address=ip_address) @implementer(IBodyProducer) @@ -456,6 +457,11 @@ class _ResponseReader(protocol.Protocol): self._fail_on_dataloss_warned = False self._reached_warnsize = False self._bytes_received = 0 + self._ip_address = None + + def connectionMade(self): + if self._ip_address is None: + self._ip_address = ip_address(self.transport._producer.getPeer().host) def dataReceived(self, bodyBytes): # This maybe called several times after cancel was called with buffered data. @@ -488,16 +494,16 @@ class _ResponseReader(protocol.Protocol): body = self._bodybuf.getvalue() if reason.check(ResponseDone): - self._finished.callback((self._txresponse, body, None)) + self._finished.callback((self._txresponse, body, None, self._ip_address)) return if reason.check(PotentialDataLoss): - self._finished.callback((self._txresponse, body, ['partial'])) + self._finished.callback((self._txresponse, body, ['partial'], self._ip_address)) return if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons): if not self._fail_on_dataloss: - self._finished.callback((self._txresponse, body, ['dataloss'])) + self._finished.callback((self._txresponse, body, ['dataloss'], self._ip_address)) return elif not self._fail_on_dataloss_warned: diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index f92d0901c..ca5ecc02c 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -17,13 +17,14 @@ from scrapy.utils.trackref import object_ref class Response(object_ref): - def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None): + def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None, ip_address=None): self.headers = Headers(headers or {}) self.status = int(status) self._set_body(body) self._set_url(url) self.request = request self.flags = [] if flags is None else list(flags) + self.ip_address = ip_address @property def meta(self): @@ -76,7 +77,7 @@ class Response(object_ref): """Create a new Response with the same attributes except for those given new values. """ - for x in ['url', 'status', 'headers', 'body', 'request', 'flags']: + for x in ['url', 'status', 'headers', 'body', 'request', 'flags', 'ip_address']: kwargs.setdefault(x, getattr(self, x)) cls = kwargs.pop('cls', self.__class__) return cls(*args, **kwargs) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index f433fcea6..6281160ae 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -1,5 +1,7 @@ import json import logging +from ipaddress import IPv4Address +from urllib.parse import urlparse from testfixtures import LogCapture from twisted.internet import defer @@ -308,3 +310,28 @@ with multiples lines self.assertIn("[callback] status 201", str(log)) self.assertIn("[errback] status 404", str(log)) self.assertIn("[errback] status 500", str(log)) + + @defer.inlineCallbacks + def test_dns_server_ip_address(self): + from socket import gethostbyname + + crawler = self.runner.create_crawler(SingleRequestSpider) + url = 'https://example.org' + yield crawler.crawl(seed=url) + ip_address = crawler.spider.meta['responses'][0].ip_address + self.assertIsInstance(ip_address, IPv4Address) + self.assertEqual(str(ip_address), gethostbyname(urlparse(url).netloc)) + + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url('/status?n=200') + yield crawler.crawl(seed=url, mockserver=self.mockserver) + ip_address = crawler.spider.meta['responses'][0].ip_address + self.assertIsNone(ip_address) + + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url('/echo?body=test') + expected_netloc, _ = urlparse(url).netloc.split(':') + yield crawler.crawl(seed=url, mockserver=self.mockserver) + ip_address = crawler.spider.meta['responses'][0].ip_address + self.assertIsInstance(ip_address, IPv4Address) + self.assertEqual(str(ip_address), gethostbyname(expected_netloc)) From e8da7e296691d2b4eb63e2a442bb600e03e5766f Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sun, 26 Jan 2020 17:53:39 -0300 Subject: [PATCH 164/313] Test DNS resolution using CrawlerProcess --- tests/CrawlerProcess/ip_address.py | 51 ++++++++++++++++++++++++++++++ tests/test_crawl.py | 10 +----- tests/test_crawler.py | 8 +++++ 3 files changed, 60 insertions(+), 9 deletions(-) create mode 100644 tests/CrawlerProcess/ip_address.py diff --git a/tests/CrawlerProcess/ip_address.py b/tests/CrawlerProcess/ip_address.py new file mode 100644 index 000000000..6b069cc90 --- /dev/null +++ b/tests/CrawlerProcess/ip_address.py @@ -0,0 +1,51 @@ +from urllib.parse import urlparse + +from twisted.internet import defer +from twisted.internet.base import ThreadedResolver +from twisted.internet.interfaces import IResolverSimple +from zope.interface.declarations import implementer + +from scrapy import Spider, Request +from scrapy.crawler import CrawlerProcess + +from tests.mockserver import MockServer + + +@implementer(IResolverSimple) +class MockThreadedResolver(ThreadedResolver): + """ + Resolves all names to localhost + """ + + @classmethod + def from_crawler(cls, crawler, reactor): + return cls(reactor) + + def install_on_reactor(self,): + self.reactor.installResolver(self) + + def getHostByName(self, name, timeout=None): + return defer.succeed("127.0.0.1") + + +class LocalhostSpider(Spider): + name = "localhost_spider" + + def start_requests(self): + yield Request(self.url) + + def parse(self, response): + netloc = urlparse(response.url).netloc + self.logger.info("Host: %s" % netloc.split(":")[0]) + self.logger.info("Type: %s" % type(response.ip_address)) + self.logger.info("IP address: %s" % response.ip_address) + + +with MockServer() as mockserver: + settings = {"DNS_RESOLVER": __name__ + ".MockThreadedResolver"} + process = CrawlerProcess(settings) + + port = urlparse(mockserver.http_address).port + url = "http://not.a.real.domain:{port}/echo?body=test".format(port=port) + process.crawl(LocalhostSpider, url=url) + process.start() diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 6281160ae..9896058dc 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -1,6 +1,7 @@ import json import logging from ipaddress import IPv4Address +from socket import gethostbyname from urllib.parse import urlparse from testfixtures import LogCapture @@ -313,15 +314,6 @@ with multiples lines @defer.inlineCallbacks def test_dns_server_ip_address(self): - from socket import gethostbyname - - crawler = self.runner.create_crawler(SingleRequestSpider) - url = 'https://example.org' - yield crawler.crawl(seed=url) - ip_address = crawler.spider.meta['responses'][0].ip_address - self.assertIsInstance(ip_address, IPv4Address) - self.assertEqual(str(ip_address), gethostbyname(urlparse(url).netloc)) - crawler = self.runner.create_crawler(SingleRequestSpider) url = self.mockserver.url('/status?n=200') yield crawler.crawl(seed=url, mockserver=self.mockserver) diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 0ce0674de..dfc1cf448 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -107,6 +107,7 @@ class CrawlerLoggingTestCase(unittest.TestCase): def test_spider_custom_settings_log_level(self): log_file = self.mktemp() + class MySpider(scrapy.Spider): name = 'spider' custom_settings = { @@ -323,3 +324,10 @@ class CrawlerProcessSubprocess(unittest.TestCase): "'downloader/exception_type_count/twisted.internet.error.ConnectionRefusedError': 1," in log, "'downloader/exception_type_count/twisted.internet.error.ConnectError': 1," in log, ])) + + def test_response_ip_address(self): + log = self.run_script("ip_address.py") + self.assertIn("Spider closed (finished)", log) + self.assertIn("Host: not.a.real.domain", log) + self.assertIn("Type: <class 'ipaddress.IPv4Address'>", log) + self.assertIn("IP address: 127.0.0.1", log) From 8529dff41d3d2f6c81ee58c60b16dd9f2b8f72b4 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sun, 26 Jan 2020 18:00:56 -0300 Subject: [PATCH 165/313] Update docs regarding Response.ip_address and IPv6 --- docs/topics/request-response.rst | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index a4cc1a7d7..17eb63064 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -612,10 +612,7 @@ Response objects :type request: :class:`Request` object :param ip_address: The IP address of the server from which the Response originated. - :type ip_address: :class:`ipaddress.IPv4Address` object - - .. FIXME: Add ipaddress.IPv6Address once it's supported - + :type ip_address: :class:`ipaddress.IPv4Address` or :class:`ipaddress.IPv6Address` .. attribute:: Response.url From c9d36522302ab73552d804137a3625552275a771 Mon Sep 17 00:00:00 2001 From: "Matsievskiy S.V" <matsievskiysv@gmail.com> Date: Mon, 27 Jan 2020 18:24:57 +0300 Subject: [PATCH 166/313] add zsh -h autocomplete option --- extras/scrapy_zsh_completion | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion index e995947cb..33f46eda8 100644 --- a/extras/scrapy_zsh_completion +++ b/extras/scrapy_zsh_completion @@ -1,11 +1,12 @@ #compdef scrapy _scrapy() { local context state state_descr line + local ret=1 typeset -A opt_args _arguments \ - "(- 1 *)--help[Help]" \ + "(- 1 *)"{-h,--help}"[Help]" \ "1: :->command" \ - "*:: :->args" + "*:: :->args" && ret=0 case $state in command) @@ -134,6 +135,8 @@ _scrapy() { esac ;; esac + + return ret } _scrapy_cmds() { From ad4477d335bee8b10bc3bbca969defddd9b316f8 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 27 Jan 2020 14:16:43 -0300 Subject: [PATCH 167/313] Remove unnecessary else --- scrapy/core/spidermw.py | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index ed02b306b..8b36cbb04 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -107,12 +107,11 @@ class SpiderMiddlewareManager(MiddlewareManager): if isinstance(exception_result, Failure): raise return exception_result + if _isiterable(result): + result = _evaluate_iterable(result, method_index+1, recovered) else: - if _isiterable(result): - result = _evaluate_iterable(result, method_index+1, recovered) - else: - msg = "Middleware {} must return an iterable, got {}" - raise _InvalidOutput(msg.format(_fname(method), type(result))) + msg = "Middleware {} must return an iterable, got {}" + raise _InvalidOutput(msg.format(_fname(method), type(result))) return MutableChain(result, recovered) From a3b168948cb533427eceb68cf84bc8542732848b Mon Sep 17 00:00:00 2001 From: Kevin Lloyd Bernal <kevinoxy@gmail.com> Date: Wed, 29 Jan 2020 04:53:25 +0800 Subject: [PATCH 168/313] Log an error when giving up requests after too many retries (#3566) --- scrapy/downloadermiddlewares/retry.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/downloadermiddlewares/retry.py b/scrapy/downloadermiddlewares/retry.py index dbc605a4c..7ab5b6e62 100644 --- a/scrapy/downloadermiddlewares/retry.py +++ b/scrapy/downloadermiddlewares/retry.py @@ -84,6 +84,6 @@ class RetryMiddleware(object): return retryreq else: stats.inc_value('retry/max_reached') - logger.debug("Gave up retrying %(request)s (failed %(retries)d times): %(reason)s", + logger.error("Gave up retrying %(request)s (failed %(retries)d times): %(reason)s", {'request': request, 'retries': retries, 'reason': reason}, extra={'spider': spider}) From 752e8f7018cbfac9cbdf486046d6bd8171cca0e8 Mon Sep 17 00:00:00 2001 From: Daniel Kimsey <dekimsey@gmail.com> Date: Sun, 26 Jan 2020 13:21:31 -0600 Subject: [PATCH 169/313] FilesPipeline.file_path has optional arguments Documented signature doesn't match the actual interface in [files.py](https://github.com/scrapy/scrapy/blob/master/scrapy/pipelines/files.py#L520). Specifically, it looks like it may be [called](https://github.com/scrapy/scrapy/blob/master/scrapy/pipelines/files.py#L422) without a response value. I found this when I was implementing the pipeline with the signature `file_path(self, request, response, info)` and the following error was being return in my results : [(False, <twisted.python.failure.Failure builtins.TypeError: file_path() missing 1 required positional argument: 'response'>)] Scrapy==1.8.0 --- docs/topics/media-pipeline.rst | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 1e0e0f18f..67a0bfdba 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -410,7 +410,7 @@ See here the methods that you can override in your custom Files Pipeline: .. class:: FilesPipeline - .. method:: file_path(request, response, info) + .. method:: file_path(self, request, response=None, info=None) This method is called once per downloaded item. It returns the download path of the file originating from the specified @@ -434,7 +434,7 @@ See here the methods that you can override in your custom Files Pipeline: class MyFilesPipeline(FilesPipeline): - def file_path(self, request, response, info): + def file_path(self, request, response=None, info=None): return 'files/' + os.path.basename(urlparse(request.url).path) By default the :meth:`file_path` method returns @@ -524,7 +524,7 @@ See here the methods that you can override in your custom Images Pipeline: The :class:`ImagesPipeline` is an extension of the :class:`FilesPipeline`, customizing the field names and adding custom behavior for images. - .. method:: file_path(request, response, info) + .. method:: file_path(self, request, response=None, info=None) This method is called once per downloaded item. It returns the download path of the file originating from the specified @@ -548,7 +548,7 @@ See here the methods that you can override in your custom Images Pipeline: class MyImagesPipeline(ImagesPipeline): - def file_path(self, request, response, info): + def file_path(self, request, response=None, info=None): return 'files/' + os.path.basename(urlparse(request.url).path) By default the :meth:`file_path` method returns From 4e56571a196c09f8976b1b31a82a8ce2d0ee0be7 Mon Sep 17 00:00:00 2001 From: Evgeny Dorofeev <evgeny@scrapinghub.com> Date: Wed, 29 Jan 2020 15:48:47 +0300 Subject: [PATCH 170/313] [HttpCompressionMiddleware] fix delimiter for Accept-Encoding header --- scrapy/downloadermiddlewares/httpcompression.py | 2 +- tests/test_downloadermiddleware_httpcompression.py | 4 ++-- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 65b652953..0010b2a8f 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -26,7 +26,7 @@ class HttpCompressionMiddleware(object): def process_request(self, request, spider): request.headers.setdefault('Accept-Encoding', - b",".join(ACCEPTED_ENCODINGS)) + b", ".join(ACCEPTED_ENCODINGS)) def process_response(self, request, response, spider): diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index c6a823b53..64488841a 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -48,7 +48,7 @@ class HttpCompressionTest(TestCase): } response = Response('http://scrapytest.org/', body=body, headers=headers) - response.request = Request('http://scrapytest.org', headers={'Accept-Encoding': 'gzip,deflate'}) + response.request = Request('http://scrapytest.org', headers={'Accept-Encoding': 'gzip, deflate'}) return response def test_process_request(self): @@ -56,7 +56,7 @@ class HttpCompressionTest(TestCase): assert 'Accept-Encoding' not in request.headers self.mw.process_request(request, self.spider) self.assertEqual(request.headers.get('Accept-Encoding'), - b','.join(ACCEPTED_ENCODINGS)) + b', '.join(ACCEPTED_ENCODINGS)) def test_process_response_gzip(self): response = self._getresponse('gzip') From cc825c21deaa56875f2abf1b30b53abb60c566c7 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 30 Jan 2020 16:17:06 +0500 Subject: [PATCH 171/313] Test returning items from an async def callback. --- tests/spiders.py | 11 +++++++++++ tests/test_crawl.py | 19 ++++++++++++++++++- 2 files changed, 29 insertions(+), 1 deletion(-) diff --git a/tests/spiders.py b/tests/spiders.py index e4f2d5474..3b1ee94b8 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -106,6 +106,17 @@ class AsyncDefAsyncioSpider(SimpleSpider): self.logger.info("Got response %d" % status) +class AsyncDefAsyncioReturnSpider(SimpleSpider): + + name = 'asyncdef_asyncio_return' + + async def parse(self, response): + await asyncio.sleep(0.2) + status = await get_from_asyncio_queue(response.status) + self.logger.info("Got response %d" % status) + return [{'id': 1}, {'id': 2}] + + class ItemSpider(FollowAllSpider): name = 'item' diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 99b887ff6..85005eba4 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -6,13 +6,14 @@ from testfixtures import LogCapture from twisted.internet import defer from twisted.trial.unittest import TestCase +from scrapy import signals from scrapy.crawler import CrawlerRunner from scrapy.http import Request from scrapy.utils.python import to_unicode from tests.mockserver import MockServer from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider, SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback, - AsyncDefSpider, AsyncDefAsyncioSpider) + AsyncDefSpider, AsyncDefAsyncioSpider, AsyncDefAsyncioReturnSpider) class CrawlTestCase(TestCase): @@ -326,3 +327,19 @@ with multiples lines with LogCapture() as log: yield runner.join() self.assertIn("Got response 200", str(log)) + + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncio_parse_list(self): + items = [] + + def _on_item_scraped(item): + items.append(item) + + crawler = self.runner.create_crawler(AsyncDefAsyncioReturnSpider) + crawler.signals.connect(_on_item_scraped, signals.item_scraped) + with LogCapture() as log: + yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + self.assertIn("Got response 200", str(log)) + self.assertIn({'id': 1}, items) + self.assertIn({'id': 2}, items) From a2ae380efcaa5a3419a4f6a35541ae0fb71a2e7f Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 3 Feb 2020 13:23:52 -0300 Subject: [PATCH 172/313] Remove unnecessary commas --- scrapy/resolver.py | 2 +- tests/CrawlerProcess/ip_address.py | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 554a3a14d..f69894b1e 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -29,7 +29,7 @@ class CachingThreadedResolver(ThreadedResolver): cache_size = 0 return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) - def install_on_reactor(self,): + def install_on_reactor(self): self.reactor.installResolver(self) def getHostByName(self, name, timeout=None): diff --git a/tests/CrawlerProcess/ip_address.py b/tests/CrawlerProcess/ip_address.py index 6b069cc90..949e97172 100644 --- a/tests/CrawlerProcess/ip_address.py +++ b/tests/CrawlerProcess/ip_address.py @@ -21,7 +21,7 @@ class MockThreadedResolver(ThreadedResolver): def from_crawler(cls, crawler, reactor): return cls(reactor) - def install_on_reactor(self,): + def install_on_reactor(self): self.reactor.installResolver(self) def getHostByName(self, name, timeout=None): From bb8f7dc609382153df79774ad9d8f6d33d064279 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 3 Feb 2020 14:50:14 -0300 Subject: [PATCH 173/313] Mock DNS server --- tests/mockserver.py | 90 ++++++++++++++++++++++++++++++++++----------- 1 file changed, 68 insertions(+), 22 deletions(-) diff --git a/tests/mockserver.py b/tests/mockserver.py index a45277db9..585741f1b 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -1,3 +1,4 @@ +import argparse import json import os import random @@ -6,18 +7,19 @@ from subprocess import Popen, PIPE from urllib.parse import urlencode from OpenSSL import SSL -from twisted.web.server import Site, NOT_DONE_YET -from twisted.web.resource import Resource +from twisted.internet import defer, reactor, ssl +from twisted.internet.task import deferLater +from twisted.names import dns, error +from twisted.names.server import DNSServerFactory +from twisted.web.resource import EncodingResourceWrapper, Resource +from twisted.web.server import GzipEncoderFactory, NOT_DONE_YET, Site from twisted.web.static import File from twisted.web.test.test_webclient import PayloadResource -from twisted.web.server import GzipEncoderFactory -from twisted.web.resource import EncodingResourceWrapper from twisted.web.util import redirectTo -from twisted.internet import reactor, ssl -from twisted.internet.task import deferLater from scrapy.utils.python import to_bytes, to_unicode from scrapy.utils.ssl import SSL_OP_NO_TLSv1_3 +from scrapy.utils.test import get_testenv def getarg(request, name, default=None, type=None): @@ -198,12 +200,10 @@ class Root(Resource): return b'Scrapy mock HTTP server\n' -class MockServer(): +class MockServer: def __enter__(self): - from scrapy.utils.test import get_testenv - - self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver'], + self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver', '-t', 'http'], stdout=PIPE, env=get_testenv()) http_address = self.proc.stdout.readline().strip().decode('ascii') https_address = self.proc.stdout.readline().strip().decode('ascii') @@ -224,6 +224,37 @@ class MockServer(): return host + path +class MockDNSResolver: + """ + Implements twisted.internet.interfaces.IResolver partially + """ + + def _resolve(self, name): + record = dns.Record_A(address=b"127.0.0.1") + answer = dns.RRHeader(name=name, payload=record) + return [answer], [], [] + + def query(self, query, timeout=None): + if query.type == dns.A: + return defer.succeed(self._resolve(query.name.name)) + return defer.fail(error.DomainError()) + + def lookupAllRecords(self, name, timeout=None): + return defer.succeed(self._resolve(name)) + + +class MockDNSServer(): + + def __enter__(self): + self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver', 'dns'], + stdout=PIPE, env=get_testenv()) + return self + + def __exit__(self, exc_type, exc_value, traceback): + self.proc.kill() + self.proc.communicate() + + def ssl_context_factory(keyfile='keys/localhost.key', certfile='keys/localhost.crt', cipher_string=None): factory = ssl.DefaultOpenSSLContextFactory( os.path.join(os.path.dirname(__file__), keyfile), @@ -238,19 +269,34 @@ def ssl_context_factory(keyfile='keys/localhost.key', certfile='keys/localhost.c if __name__ == "__main__": - root = Root() - factory = Site(root) - httpPort = reactor.listenTCP(0, factory) - contextFactory = ssl_context_factory() - httpsPort = reactor.listenSSL(0, factory, contextFactory) + parser = argparse.ArgumentParser() + parser.add_argument("-t", "--type", type=str, choices=("http", "dns"), default="http") + args = parser.parse_args() - def print_listening(): - httpHost = httpPort.getHost() - httpsHost = httpsPort.getHost() - httpAddress = 'http://%s:%d' % (httpHost.host, httpHost.port) - httpsAddress = 'https://%s:%d' % (httpsHost.host, httpsHost.port) - print(httpAddress) - print(httpsAddress) + if args.type == "http": + root = Root() + factory = Site(root) + httpPort = reactor.listenTCP(0, factory) + contextFactory = ssl_context_factory() + httpsPort = reactor.listenSSL(0, factory, contextFactory) + + def print_listening(): + httpHost = httpPort.getHost() + httpsHost = httpsPort.getHost() + httpAddress = "http://%s:%d" % (httpHost.host, httpHost.port) + httpsAddress = "https://%s:%d" % (httpsHost.host, httpsHost.port) + print(httpAddress) + print(httpsAddress) + + elif args.type == "dns": + clients = [MockDNSResolver()] + factory = DNSServerFactory(clients=clients) + protocol = dns.DNSDatagramProtocol(controller=factory) + reactor.listenUDP(10053, protocol) + reactor.listenTCP(10053, factory) + + def print_listening(): + print("DNS server running on port 10053") reactor.callWhenRunning(print_listening) reactor.run() From 4851efdfb0885a40a44a2834c6c69d0104326801 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 3 Feb 2020 14:50:54 -0300 Subject: [PATCH 174/313] Flake8 adjustments --- tests/mockserver.py | 6 +++--- 1 file changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/mockserver.py b/tests/mockserver.py index 585741f1b..67139534e 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -257,9 +257,9 @@ class MockDNSServer(): def ssl_context_factory(keyfile='keys/localhost.key', certfile='keys/localhost.crt', cipher_string=None): factory = ssl.DefaultOpenSSLContextFactory( - os.path.join(os.path.dirname(__file__), keyfile), - os.path.join(os.path.dirname(__file__), certfile), - ) + os.path.join(os.path.dirname(__file__), keyfile), + os.path.join(os.path.dirname(__file__), certfile), + ) if cipher_string: ctx = factory.getContext() # disabling TLS1.2+ because it unconditionally enables some strong ciphers From e0ef8ad2d6f958de6ce04cd7756e142efeb1a6a2 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 3 Feb 2020 15:52:15 -0300 Subject: [PATCH 175/313] CrawlerRunner test for Response.ip_address --- tests/CrawlerProcess/ip_address.py | 51 ------------------------------ tests/CrawlerRunner/ip_address.py | 37 ++++++++++++++++++++++ tests/mockserver.py | 11 ++++--- tests/test_crawler.py | 20 ++++++++---- 4 files changed, 57 insertions(+), 62 deletions(-) delete mode 100644 tests/CrawlerProcess/ip_address.py create mode 100644 tests/CrawlerRunner/ip_address.py diff --git a/tests/CrawlerProcess/ip_address.py b/tests/CrawlerProcess/ip_address.py deleted file mode 100644 index 949e97172..000000000 --- a/tests/CrawlerProcess/ip_address.py +++ /dev/null @@ -1,51 +0,0 @@ -from urllib.parse import urlparse - -from twisted.internet import defer -from twisted.internet.base import ThreadedResolver -from twisted.internet.interfaces import IResolverSimple -from zope.interface.declarations import implementer - -from scrapy import Spider, Request -from scrapy.crawler import CrawlerProcess - -from tests.mockserver import MockServer - - -@implementer(IResolverSimple) -class MockThreadedResolver(ThreadedResolver): - """ - Resolves all names to localhost - """ - - @classmethod - def from_crawler(cls, crawler, reactor): - return cls(reactor) - - def install_on_reactor(self): - self.reactor.installResolver(self) - - def getHostByName(self, name, timeout=None): - return defer.succeed("127.0.0.1") - - -class LocalhostSpider(Spider): - name = "localhost_spider" - - def start_requests(self): - yield Request(self.url) - - def parse(self, response): - netloc = urlparse(response.url).netloc - self.logger.info("Host: %s" % netloc.split(":")[0]) - self.logger.info("Type: %s" % type(response.ip_address)) - self.logger.info("IP address: %s" % response.ip_address) - - -with MockServer() as mockserver: - settings = {"DNS_RESOLVER": __name__ + ".MockThreadedResolver"} - process = CrawlerProcess(settings) - - port = urlparse(mockserver.http_address).port - url = "http://not.a.real.domain:{port}/echo?body=test".format(port=port) - process.crawl(LocalhostSpider, url=url) - process.start() diff --git a/tests/CrawlerRunner/ip_address.py b/tests/CrawlerRunner/ip_address.py new file mode 100644 index 000000000..5a71536d8 --- /dev/null +++ b/tests/CrawlerRunner/ip_address.py @@ -0,0 +1,37 @@ +from urllib.parse import urlparse + +from twisted.internet import reactor +from twisted.names.client import createResolver + +from scrapy import Spider, Request +from scrapy.crawler import CrawlerRunner +from scrapy.utils.log import configure_logging + +from tests.mockserver import MockServer, MockDNSServer + + +class LocalhostSpider(Spider): + name = "localhost_spider" + + def start_requests(self): + yield Request(self.url) + + def parse(self, response): + netloc = urlparse(response.url).netloc + self.logger.info("Host: %s" % netloc.split(":")[0]) + self.logger.info("Type: %s" % type(response.ip_address)) + self.logger.info("IP address: %s" % response.ip_address) + + +with MockServer() as mock_http_server, MockDNSServer() as mock_dns_server: + port = urlparse(mock_http_server.http_address).port + url = "http://not.a.real.domain:{port}/echo".format(port=port) + + servers = [(mock_dns_server.host, mock_dns_server.port)] + reactor.installResolver(createResolver(servers=servers)) + + configure_logging() + runner = CrawlerRunner() + d = runner.crawl(LocalhostSpider, url=url) + d.addBoth(lambda _: reactor.stop()) + reactor.run() diff --git a/tests/mockserver.py b/tests/mockserver.py index 67139534e..08a81418c 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -246,8 +246,11 @@ class MockDNSResolver: class MockDNSServer(): def __enter__(self): - self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver', 'dns'], + self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver', '-t', 'dns'], stdout=PIPE, env=get_testenv()) + host, port = self.proc.stdout.readline().strip().decode('ascii').split(":") + self.host = host + self.port = int(port) return self def __exit__(self, exc_type, exc_value, traceback): @@ -292,11 +295,11 @@ if __name__ == "__main__": clients = [MockDNSResolver()] factory = DNSServerFactory(clients=clients) protocol = dns.DNSDatagramProtocol(controller=factory) - reactor.listenUDP(10053, protocol) - reactor.listenTCP(10053, factory) + listener = reactor.listenUDP(0, protocol) def print_listening(): - print("DNS server running on port 10053") + host = listener.getHost() + print("%s:%s" % (host.host, host.port)) reactor.callWhenRunning(print_listening) reactor.run() diff --git a/tests/test_crawler.py b/tests/test_crawler.py index dfc1cf448..5d381c368 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -281,9 +281,7 @@ class CrawlerRunnerHasSpider(unittest.TestCase): self.assertNotIn("Asyncio reactor is installed", str(log)) -class CrawlerProcessSubprocess(unittest.TestCase): - script_dir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'CrawlerProcess') - +class ScriptRunnerMixin: def run_script(self, script_name): script_path = os.path.join(self.script_dir, script_name) args = (sys.executable, script_path) @@ -292,6 +290,10 @@ class CrawlerProcessSubprocess(unittest.TestCase): stdout, stderr = p.communicate() return stderr.decode('utf-8') + +class CrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase): + script_dir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'CrawlerProcess') + def test_simple(self): log = self.run_script('simple.py') self.assertIn('Spider closed (finished)', log) @@ -325,9 +327,13 @@ class CrawlerProcessSubprocess(unittest.TestCase): "'downloader/exception_type_count/twisted.internet.error.ConnectError': 1," in log, ])) + +class CrawlerRunnerSubprocess(ScriptRunnerMixin, unittest.TestCase): + script_dir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'CrawlerRunner') + def test_response_ip_address(self): log = self.run_script("ip_address.py") - self.assertIn("Spider closed (finished)", log) - self.assertIn("Host: not.a.real.domain", log) - self.assertIn("Type: <class 'ipaddress.IPv4Address'>", log) - self.assertIn("IP address: 127.0.0.1", log) + self.assertIn("INFO: Spider closed (finished)", log) + self.assertIn("INFO: Host: not.a.real.domain", log) + self.assertIn("INFO: Type: <class 'ipaddress.IPv4Address'>", log) + self.assertIn("INFO: IP address: 127.0.0.1", log) From 13670f0397ba8dcec3dceb1852bad5751406d19d Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 3 Feb 2020 16:16:43 -0300 Subject: [PATCH 176/313] Ignore tests/CrawlerRunner directory --- conftest.py | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/conftest.py b/conftest.py index c0de09909..55294feca 100644 --- a/conftest.py +++ b/conftest.py @@ -11,7 +11,8 @@ collect_ignore = [ # not a test, but looks like a test "scrapy/utils/testsite.py", # contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess - *_py_files("tests/CrawlerProcess") + *_py_files("tests/CrawlerProcess"), + *_py_files("tests/CrawlerRunner"), ] for line in open('tests/ignores.txt'): From ad70497416527c3d882a64f7803e73155f3fa1da Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Tue, 4 Feb 2020 13:30:13 -0300 Subject: [PATCH 177/313] Remove unnecessary parentheses in class definition --- tests/mockserver.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/mockserver.py b/tests/mockserver.py index 08a81418c..30d9bc0e8 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -243,7 +243,7 @@ class MockDNSResolver: return defer.succeed(self._resolve(name)) -class MockDNSServer(): +class MockDNSServer: def __enter__(self): self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver', '-t', 'dns'], From fbea370c58c1d82b52fd9c1f7d3a6cee94477c7a Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 5 Feb 2020 01:35:13 -0300 Subject: [PATCH 178/313] Rename function parameter --- scrapy/core/spidermw.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index 8b36cbb04..dd9b3c376 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -59,12 +59,12 @@ class SpiderMiddlewareManager(MiddlewareManager): return scrape_func(Failure(), request, spider) return scrape_func(response, request, spider) - def _evaluate_iterable(iterable, method_index, recover_to): + def _evaluate_iterable(iterable, exception_processor_index, recover_to): try: for r in iterable: yield r except Exception as ex: - exception_result = process_spider_exception(Failure(ex), method_index) + exception_result = process_spider_exception(Failure(ex), exception_processor_index) if isinstance(exception_result, Failure): raise recover_to.extend(exception_result) From 898bc00811aac9d3e38d1863b95a10c2e8effb02 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 5 Feb 2020 11:31:27 +0000 Subject: [PATCH 179/313] new signal --- scrapy/signals.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scrapy/signals.py b/scrapy/signals.py index 6b9125302..cd7ed7fb1 100644 --- a/scrapy/signals.py +++ b/scrapy/signals.py @@ -14,6 +14,7 @@ spider_error = object() request_scheduled = object() request_dropped = object() request_reached_downloader = object() +request_left_downloader = object() response_received = object() response_downloaded = object() item_scraped = object() From ae04174884eeb777d7b3caceed52bf522944ceb1 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 5 Feb 2020 11:32:31 +0000 Subject: [PATCH 180/313] emit new signal --- scrapy/core/downloader/__init__.py | 3 +++ 1 file changed, 3 insertions(+) diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index 157dc3418..5a2fdadf5 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -181,6 +181,9 @@ class Downloader(object): def finish_transferring(_): slot.transferring.remove(request) self._process_queue(spider, slot) + self.signals.send_catch_log(signal=signals.request_left_downloader, + request=request, + spider=spider) return _ return dfd.addBoth(finish_transferring) From 9916f6e556f9d4a41ea86d4a73687af1a40e43ba Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 5 Feb 2020 11:32:54 +0000 Subject: [PATCH 181/313] tests for new signal --- tests/test_request_left.py | 59 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 59 insertions(+) create mode 100644 tests/test_request_left.py diff --git a/tests/test_request_left.py b/tests/test_request_left.py new file mode 100644 index 000000000..ddeca0499 --- /dev/null +++ b/tests/test_request_left.py @@ -0,0 +1,59 @@ +from twisted.internet import defer +from twisted.trial.unittest import TestCase +from scrapy.signals import request_left_downloader +from scrapy.spiders import Spider +from scrapy.utils.test import get_crawler +from tests.mockserver import MockServer + +class SignalCatcherSpider(Spider): + name = 'signal_catcher' + + def __init__(self, crawler, url, *args, **kwargs): + super(SignalCatcherSpider, self).__init__(*args, **kwargs) + crawler.signals.connect(self.on_response_download, + signal=request_left_downloader) + self.catched_times = 0 + self.start_urls = [url] + + @classmethod + def from_crawler(cls, crawler, *args, **kwargs): + spider = cls(crawler, *args, **kwargs) + return spider + + def on_response_download(self, request, spider): + self.catched_times = self.catched_times + 1 + + +class TestCatching(TestCase): + + def setUp(self): + self.mockserver = MockServer() + self.mockserver.__enter__() + + def tearDown(self): + self.mockserver.__exit__(None, None, None) + + @defer.inlineCallbacks + def test_success(self): + crawler = get_crawler(SignalCatcherSpider) + yield crawler.crawl(self.mockserver.url("/status?n=200")) + self.assertEqual(crawler.spider.catched_times, 1) + + @defer.inlineCallbacks + def test_timeout(self): + crawler = get_crawler(SignalCatcherSpider, + {'DOWNLOAD_TIMEOUT': 0.1}) + yield crawler.crawl(self.mockserver.url("/delay?n=0.2")) + self.assertEqual(crawler.spider.catched_times, 1) + + @defer.inlineCallbacks + def test_disconnect(self): + crawler = get_crawler(SignalCatcherSpider) + yield crawler.crawl(self.mockserver.url("/drop")) + self.assertEqual(crawler.spider.catched_times, 1) + + @defer.inlineCallbacks + def test_noconnect(self): + crawler = get_crawler(SignalCatcherSpider) + yield crawler.crawl('http://thereisdefinetelynosuchdomain.com') + self.assertEqual(crawler.spider.catched_times, 1) From aab39f63412b4b7a0ae2713446859d6d8103e5f7 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 5 Feb 2020 11:35:03 +0000 Subject: [PATCH 182/313] docummentation for new signal --- docs/topics/signals.rst | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 3f29aa323..7fa5bc030 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -295,6 +295,23 @@ request_reached_downloader :param spider: the spider that yielded the request :type spider: :class:`~scrapy.spiders.Spider` object +request_left_downloader +--------------------------- + +.. signal:: request_left_downloader +.. function:: request_left_downloader(request, spider) + + Sent when a :class:`~scrapy.http.Request` left downloader even in case of + failure. + + The signal does not support returning deferreds from their handlers. + + :param request: the request that reached downloader + :type request: :class:`~scrapy.http.Request` object + + :param spider: the spider that yielded the request + :type spider: :class:`~scrapy.spiders.Spider` object + response_received ----------------- From 3769f75386104c1a3072894b302d3c3239ff8c37 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Wed, 5 Feb 2020 12:08:08 +0000 Subject: [PATCH 183/313] pep8 E302 --- tests/test_request_left.py | 1 + 1 file changed, 1 insertion(+) diff --git a/tests/test_request_left.py b/tests/test_request_left.py index ddeca0499..5d271190d 100644 --- a/tests/test_request_left.py +++ b/tests/test_request_left.py @@ -5,6 +5,7 @@ from scrapy.spiders import Spider from scrapy.utils.test import get_crawler from tests.mockserver import MockServer + class SignalCatcherSpider(Spider): name = 'signal_catcher' From 11941c324431e1f8822e64fb03d163b9e721eaa5 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 5 Feb 2020 13:27:54 -0300 Subject: [PATCH 184/313] Remove elusive six occurrence from tox.ini --- tox.ini | 1 - 1 file changed, 1 deletion(-) diff --git a/tox.ini b/tox.ini index b62100026..b1babc7fd 100644 --- a/tox.ini +++ b/tox.ini @@ -56,7 +56,6 @@ deps = pyOpenSSL==16.2.0 queuelib==1.4.2 service_identity==16.0.0 - six==1.10.0 Twisted==17.9.0 w3lib==1.17.0 zope.interface==4.1.3 From c2cca368213013f9dc9d4569b926da4d213f20a5 Mon Sep 17 00:00:00 2001 From: Respawnz <47511522+Respawnz@users.noreply.github.com> Date: Thu, 6 Feb 2020 05:39:15 +0800 Subject: [PATCH 185/313] typo --- docs/topics/developer-tools.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index bf14643be..1fedf91df 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -132,7 +132,7 @@ a use case: Say you want to find the ``Next`` button on the page. Type ``Next`` into the search bar on the top right of the `Inspector`. You should get two results. -The first is a ``li`` tag with the ``class="text"``, the second the text +The first is a ``li`` tag with the ``class="next"``, the second the text of an ``a`` tag. Right click on the ``a`` tag and select ``Scroll into View``. If you hover over the tag, you'll see the button highlighted. From here we could easily create a :ref:`Link Extractor <topics-link-extractors>` to From 576663e5a778646aa4ef870641fa677ea94d21f9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 6 Feb 2020 10:43:20 +0100 Subject: [PATCH 186/313] Make METAREFRESH_IGNORE_TAGS an empty list by default --- docs/topics/downloader-middleware.rst | 2 +- scrapy/settings/default_settings.py | 2 +- tests/test_downloadermiddleware_redirect.py | 16 +++++++++------- 3 files changed, 11 insertions(+), 9 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 8a760e53b..3ec6e0c17 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -868,7 +868,7 @@ Whether the Meta Refresh middleware will be enabled. METAREFRESH_IGNORE_TAGS ^^^^^^^^^^^^^^^^^^^^^^^ -Default: ``['script', 'noscript']`` +Default: ``[]`` Meta tags within these tags are ignored. diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index c10dc1a1c..1a7d35b13 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -225,7 +225,7 @@ MEMUSAGE_NOTIFY_MAIL = [] MEMUSAGE_WARNING_MB = 0 METAREFRESH_ENABLED = True -METAREFRESH_IGNORE_TAGS = ['script', 'noscript'] +METAREFRESH_IGNORE_TAGS = [] METAREFRESH_MAXDELAY = 100 NEWSPIDER_MODULE = '' diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index e7faf14a7..e0f145d0e 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -300,19 +300,21 @@ class MetaRefreshMiddlewareTest(unittest.TestCase): body = ('''<noscript><meta http-equiv="refresh" ''' '''content="0;URL='http://example.org/newpage'"></noscript>''') rsp = HtmlResponse(req.url, body=body.encode()) - response = self.mw.process_response(req, rsp, self.spider) - assert isinstance(response, Response) + req2 = self.mw.process_response(req, rsp, self.spider) + assert isinstance(req2, Request) + self.assertEqual(req2.url, 'http://example.org/newpage') - def test_ignore_tags_empty_list(self): - crawler = get_crawler(Spider, {'METAREFRESH_IGNORE_TAGS': []}) + def test_ignore_tags_1_x_list(self): + """Test that Scrapy 1.x behavior remains possible""" + settings = {'METAREFRESH_IGNORE_TAGS': ['script', 'noscript']} + crawler = get_crawler(Spider, settings) mw = MetaRefreshMiddleware.from_crawler(crawler) req = Request(url='http://example.org') body = ('''<noscript><meta http-equiv="refresh" ''' '''content="0;URL='http://example.org/newpage'"></noscript>''') rsp = HtmlResponse(req.url, body=body.encode()) - req2 = mw.process_response(req, rsp, self.spider) - assert isinstance(req2, Request) - self.assertEqual(req2.url, 'http://example.org/newpage') + response = mw.process_response(req, rsp, self.spider) + assert isinstance(response, Response) if __name__ == "__main__": unittest.main() From 6733f4d976150e0e5352d4ae9697880ae60ad638 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Thu, 6 Feb 2020 18:40:42 +0500 Subject: [PATCH 187/313] Update docs/topics/signals.rst Co-Authored-By: elacuesta <elacuesta@users.noreply.github.com> --- docs/topics/signals.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 7fa5bc030..47be6b603 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -301,7 +301,7 @@ request_left_downloader .. signal:: request_left_downloader .. function:: request_left_downloader(request, spider) - Sent when a :class:`~scrapy.http.Request` left downloader even in case of + Sent when a :class:`~scrapy.http.Request` leaves the downloader even in case of failure. The signal does not support returning deferreds from their handlers. From 4a91a5427df4846ed9fa11612cfeb9e31f34a1c8 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Thu, 6 Feb 2020 13:44:51 +0000 Subject: [PATCH 188/313] fix typo --- tests/test_request_left.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/test_request_left.py b/tests/test_request_left.py index 5d271190d..8256d1c92 100644 --- a/tests/test_request_left.py +++ b/tests/test_request_left.py @@ -13,7 +13,7 @@ class SignalCatcherSpider(Spider): super(SignalCatcherSpider, self).__init__(*args, **kwargs) crawler.signals.connect(self.on_response_download, signal=request_left_downloader) - self.catched_times = 0 + self.caught_times = 0 self.start_urls = [url] @classmethod @@ -22,7 +22,7 @@ class SignalCatcherSpider(Spider): return spider def on_response_download(self, request, spider): - self.catched_times = self.catched_times + 1 + self.caught_times = self.caught_times + 1 class TestCatching(TestCase): @@ -38,23 +38,23 @@ class TestCatching(TestCase): def test_success(self): crawler = get_crawler(SignalCatcherSpider) yield crawler.crawl(self.mockserver.url("/status?n=200")) - self.assertEqual(crawler.spider.catched_times, 1) + self.assertEqual(crawler.spider.caught_times, 1) @defer.inlineCallbacks def test_timeout(self): crawler = get_crawler(SignalCatcherSpider, {'DOWNLOAD_TIMEOUT': 0.1}) yield crawler.crawl(self.mockserver.url("/delay?n=0.2")) - self.assertEqual(crawler.spider.catched_times, 1) + self.assertEqual(crawler.spider.caught_times, 1) @defer.inlineCallbacks def test_disconnect(self): crawler = get_crawler(SignalCatcherSpider) yield crawler.crawl(self.mockserver.url("/drop")) - self.assertEqual(crawler.spider.catched_times, 1) + self.assertEqual(crawler.spider.caught_times, 1) @defer.inlineCallbacks def test_noconnect(self): crawler = get_crawler(SignalCatcherSpider) yield crawler.crawl('http://thereisdefinetelynosuchdomain.com') - self.assertEqual(crawler.spider.catched_times, 1) + self.assertEqual(crawler.spider.caught_times, 1) From 4be19e443e9c101a248c21509ae8000ce500d51a Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Thu, 6 Feb 2020 13:46:23 +0000 Subject: [PATCH 189/313] name signla catcher in accord with signal name --- tests/test_request_left.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/test_request_left.py b/tests/test_request_left.py index 8256d1c92..5cfef8e7d 100644 --- a/tests/test_request_left.py +++ b/tests/test_request_left.py @@ -11,7 +11,7 @@ class SignalCatcherSpider(Spider): def __init__(self, crawler, url, *args, **kwargs): super(SignalCatcherSpider, self).__init__(*args, **kwargs) - crawler.signals.connect(self.on_response_download, + crawler.signals.connect(self.on_request_left, signal=request_left_downloader) self.caught_times = 0 self.start_urls = [url] @@ -21,7 +21,7 @@ class SignalCatcherSpider(Spider): spider = cls(crawler, *args, **kwargs) return spider - def on_response_download(self, request, spider): + def on_request_left(self, request, spider): self.caught_times = self.caught_times + 1 From 489ffcda5143a2ef28d4cbcf5418babd963f2b0f Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Thu, 6 Feb 2020 22:39:00 +0500 Subject: [PATCH 190/313] Add a test for an async item_scraped handler. --- tests/test_signals.py | 39 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 39 insertions(+) create mode 100644 tests/test_signals.py diff --git a/tests/test_signals.py b/tests/test_signals.py new file mode 100644 index 000000000..001e798e5 --- /dev/null +++ b/tests/test_signals.py @@ -0,0 +1,39 @@ +from twisted.internet import defer +from twisted.trial import unittest + +from scrapy import signals, Request, Spider +from scrapy.utils.test import get_crawler + +from tests.mockserver import MockServer + + +class ItemSpider(Spider): + name = 'itemspider' + + def start_requests(self): + for _ in range(10): + yield Request(self.mockserver.url('/status?n=200'), + dont_filter=True) + + def parse(self, response): + return {'field': 42} + + +class AsyncSignalTestCase(unittest.TestCase): + def setUp(self): + self.mockserver = MockServer() + self.mockserver.__enter__() + self.items = [] + + def tearDown(self): + self.mockserver.__exit__(None, None, None) + + async def _on_item_scraped(self, item): + self.items.append(item) + + @defer.inlineCallbacks + def test_simple_pipeline(self): + crawler = get_crawler(ItemSpider) + crawler.signals.connect(self._on_item_scraped, signals.item_scraped) + yield crawler.crawl(mockserver=self.mockserver) + self.assertEqual(len(self.items), 10) From 35dafef7f106dbf1c022d997b4e29e3eee84de7c Mon Sep 17 00:00:00 2001 From: elacuesta <elacuesta@users.noreply.github.com> Date: Thu, 6 Feb 2020 14:42:34 -0300 Subject: [PATCH 191/313] Specify Twisted reactor (TWISTED_REACTOR setting) (#4294) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Add the ability to install a specific reactor * Add docs for the TWISTED_REACTOR setting * Add tests for the TWISTED_REACTOR setting * Update asyncio reactor test * Ignore W503 globally W503 is not PEP8-compliant: https://github.com/python/peps/commit/c59c4376ad233a62ca4b3a6060c81368bd21e85b * Line length adjustment * Adjust asyncio reactor tests * Merge ASYNCIO_ENABLED and TWISTED_REACTOR settings * More docs about TWISTED_REACTOR * Fix asyncio reactor test * Docs: fix title * Reword docs * Check the TWISTED_REACTOR setting outside of the installing function * Remove unrelated change * Update scrapy/utils/log.py Co-Authored-By: Adrián Chaves <adrian@chaves.io> * Update docs/topics/settings.rst Co-Authored-By: Adrián Chaves <adrian@chaves.io> * Update docs/topics/settings.rst Co-Authored-By: Adrián Chaves <adrian@chaves.io> Co-authored-by: Adrián Chaves <adrian@chaves.io> --- docs/faq.rst | 11 +++++ docs/topics/broad-crawls.rst | 7 +++ docs/topics/settings.rst | 45 +++++++++--------- pytest.ini | 7 +-- scrapy/crawler.py | 19 ++++---- scrapy/settings/default_settings.py | 4 +- scrapy/utils/asyncio.py | 17 ------- scrapy/utils/defer.py | 4 +- scrapy/utils/log.py | 11 ++--- scrapy/utils/reactor.py | 35 +++++++++++++- .../asyncio_enabled_no_reactor.py | 3 +- .../CrawlerProcess/asyncio_enabled_reactor.py | 3 +- .../CrawlerProcess/twisted_reactor_asyncio.py | 13 +++++ tests/CrawlerProcess/twisted_reactor_poll.py | 13 +++++ .../CrawlerProcess/twisted_reactor_select.py | 13 +++++ tests/test_commands.py | 10 ++-- tests/test_crawler.py | 47 ++++++++++++++----- tests/test_utils_asyncio.py | 4 +- 18 files changed, 182 insertions(+), 84 deletions(-) delete mode 100644 scrapy/utils/asyncio.py create mode 100644 tests/CrawlerProcess/twisted_reactor_asyncio.py create mode 100644 tests/CrawlerProcess/twisted_reactor_poll.py create mode 100644 tests/CrawlerProcess/twisted_reactor_select.py diff --git a/docs/faq.rst b/docs/faq.rst index b65908012..f72e4cf01 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -361,6 +361,17 @@ Note that by doing so, you lose the ability to set a specific timeout for DNS re (the value of the :setting:`DNS_TIMEOUT` setting is ignored). +.. _faq-specific-reactor: + +How to deal with ``<class 'ValueError'>: filedescriptor out of range in select()`` exceptions? +---------------------------------------------------------------------------------------------- + +This issue `has been reported`_ to appear when running broad crawls in macOS, where the default +Twisted reactor is :class:`twisted.internet.selectreactor.SelectReactor`. Switching to a +different reactor is possible by using the :setting:`TWISTED_REACTOR` setting. + + +.. _has been reported: https://github.com/scrapy/scrapy/issues/2905 .. _user agents: https://en.wikipedia.org/wiki/User_agent .. _LIFO: https://en.wikipedia.org/wiki/Stack_(abstract_data_type) .. _DFO order: https://en.wikipedia.org/wiki/Depth-first_search diff --git a/docs/topics/broad-crawls.rst b/docs/topics/broad-crawls.rst index 1ab08d949..4922694ee 100644 --- a/docs/topics/broad-crawls.rst +++ b/docs/topics/broad-crawls.rst @@ -211,3 +211,10 @@ If your broad crawl shows a high memory usage, in addition to :ref:`crawling in BFO order <broad-crawls-bfo>` and :ref:`lowering concurrency <broad-crawls-concurrency>` you should :ref:`debug your memory leaks <topics-leaks>`. + + +Install a specific Twisted reactor +================================== + +If the crawl is exceeding the system's capabilities, you might want to try +installing a specific Twisted reactor, via the :setting:`TWISTED_REACTOR` setting. diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 4b770d249..fa63a5807 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -160,27 +160,6 @@ to any particular component. In that case the module of that component will be shown, typically an extension, middleware or pipeline. It also means that the component must be enabled in order for the setting to have any effect. -.. setting:: ASYNCIO_REACTOR - -ASYNCIO_REACTOR ---------------- - -Default: ``False`` - -Whether to install and require the Twisted reactor that uses the asyncio loop. - -When this option is set to ``True``, Scrapy will require -:class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`. It will -install this reactor if no reactor is installed yet, such as when using the -``scrapy`` script or :class:`~scrapy.crawler.CrawlerProcess`. If you are using -:class:`~scrapy.crawler.CrawlerRunner`, you need to install the correct reactor -manually. If a different reactor is installed outside Scrapy, it will raise an -exception. - -The default value for this option is currently ``False`` to maintain backward -compatibility and avoid possible problems caused by using a different Twisted -reactor. - .. setting:: AWS_ACCESS_KEY_ID AWS_ACCESS_KEY_ID @@ -1463,6 +1442,30 @@ command. The project name must not conflict with the name of custom files or directories in the ``project`` subdirectory. +.. setting:: TWISTED_REACTOR + +TWISTED_REACTOR +--------------- + +Default: ``None`` + +Import path of a given Twisted reactor, for instance: +:class:`twisted.internet.asyncioreactor.AsyncioSelectorReactor`. + +Scrapy will install this reactor if no other is installed yet, such as when +the ``scrapy`` CLI program is invoked or when using the +:class:`~scrapy.crawler.CrawlerProcess` class. If you are using the +:class:`~scrapy.crawler.CrawlerRunner` class, you need to install the correct +reactor manually. An exception will be raised if the installation fails. + +The default value for this option is currently ``None``, which means that Scrapy +will not attempt to install any specific reactor, and the default one defined by +Twisted for the current platform will be used. This is to maintain backward +compatibility and avoid possible problems caused by using a non-default reactor. + +For additional information, please see +:doc:`core/howto/choosing-reactor`. + .. setting:: URLLENGTH_LIMIT diff --git a/pytest.ini b/pytest.ini index bae68cd3a..552829d4e 100644 --- a/pytest.ini +++ b/pytest.ini @@ -21,6 +21,7 @@ twisted = 1 markers = only_asyncio: marks tests as only enabled when --reactor=asyncio is passed flake8-ignore = + W503 # Files that are only meant to provide top-level imports are expected not # to use any of their imports: scrapy/core/downloader/handlers/http.py F401 @@ -109,7 +110,7 @@ flake8-ignore = # scrapy/spidermiddlewares scrapy/spidermiddlewares/httperror.py E501 scrapy/spidermiddlewares/offsite.py E501 - scrapy/spidermiddlewares/referer.py E501 E129 W503 W504 + scrapy/spidermiddlewares/referer.py E501 E129 W504 scrapy/spidermiddlewares/urllength.py E501 # scrapy/spiders scrapy/spiders/__init__.py E501 E402 @@ -129,13 +130,13 @@ flake8-ignore = scrapy/utils/http.py F403 E226 scrapy/utils/httpobj.py E501 scrapy/utils/iterators.py E501 E701 - scrapy/utils/log.py E128 W503 + scrapy/utils/log.py E128 E501 scrapy/utils/markup.py F403 scrapy/utils/misc.py E501 E226 scrapy/utils/multipart.py F403 scrapy/utils/project.py E501 scrapy/utils/python.py E501 - scrapy/utils/reactor.py E226 + scrapy/utils/reactor.py E226 E501 scrapy/utils/reqser.py E501 scrapy/utils/request.py E127 E501 scrapy/utils/response.py E501 E128 diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 35c6b7716..49b8e4511 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -13,7 +13,6 @@ from scrapy.extension import ExtensionManager from scrapy.interfaces import ISpiderLoader from scrapy.settings import overridden_settings, Settings from scrapy.signalmanager import SignalManager -from scrapy.utils.asyncio import install_asyncio_reactor, is_asyncio_reactor_installed from scrapy.utils.log import ( configure_logging, get_scrapy_root_handler, @@ -23,6 +22,7 @@ from scrapy.utils.log import ( ) from scrapy.utils.misc import create_instance, load_object from scrapy.utils.ossignal import install_shutdown_handlers, signal_names +from scrapy.utils.reactor import install_reactor, verify_installed_reactor logger = logging.getLogger(__name__) @@ -138,7 +138,7 @@ class CrawlerRunner: self._crawlers = set() self._active = set() self.bootstrap_failed = False - self._handle_asyncio_reactor() + self._handle_twisted_reactor() @property def spiders(self): @@ -232,10 +232,9 @@ class CrawlerRunner: while self._active: yield defer.DeferredList(self._active) - def _handle_asyncio_reactor(self): - if self.settings.getbool('ASYNCIO_REACTOR') and not is_asyncio_reactor_installed(): - raise Exception("ASYNCIO_REACTOR is on but the Twisted asyncio " - "reactor is not installed.") + def _handle_twisted_reactor(self): + if self.settings.get("TWISTED_REACTOR"): + verify_installed_reactor(self.settings["TWISTED_REACTOR"]) class CrawlerProcess(CrawlerRunner): @@ -324,10 +323,10 @@ class CrawlerProcess(CrawlerRunner): except RuntimeError: # raised if already stopped or in shutdown stage pass - def _handle_asyncio_reactor(self): - if self.settings.getbool('ASYNCIO_REACTOR'): - install_asyncio_reactor() - super()._handle_asyncio_reactor() + def _handle_twisted_reactor(self): + if self.settings.get("TWISTED_REACTOR"): + install_reactor(self.settings["TWISTED_REACTOR"]) + super()._handle_twisted_reactor() def _get_spider_loader(settings): diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index c10dc1a1c..c80833710 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -19,8 +19,6 @@ from os.path import join, abspath, dirname AJAXCRAWL_ENABLED = False -ASYNCIO_REACTOR = False - AUTOTHROTTLE_ENABLED = False AUTOTHROTTLE_DEBUG = False AUTOTHROTTLE_MAX_DELAY = 60.0 @@ -291,6 +289,8 @@ TELNETCONSOLE_HOST = '127.0.0.1' TELNETCONSOLE_USERNAME = 'scrapy' TELNETCONSOLE_PASSWORD = None +TWISTED_REACTOR = None + SPIDER_CONTRACTS = {} SPIDER_CONTRACTS_BASE = { 'scrapy.contracts.default.UrlContract': 1, diff --git a/scrapy/utils/asyncio.py b/scrapy/utils/asyncio.py deleted file mode 100644 index 917973de2..000000000 --- a/scrapy/utils/asyncio.py +++ /dev/null @@ -1,17 +0,0 @@ -import asyncio -from contextlib import suppress - -from twisted.internet import asyncioreactor -from twisted.internet.error import ReactorAlreadyInstalledError - - -def install_asyncio_reactor(): - """ Tries to install AsyncioSelectorReactor - """ - with suppress(ReactorAlreadyInstalledError): - asyncioreactor.install(asyncio.get_event_loop()) - - -def is_asyncio_reactor_installed(): - from twisted.internet import reactor - return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor) diff --git a/scrapy/utils/defer.py b/scrapy/utils/defer.py index 62b43a96c..6a21a4903 100644 --- a/scrapy/utils/defer.py +++ b/scrapy/utils/defer.py @@ -2,14 +2,14 @@ Helper functions for dealing with Twisted deferreds """ import asyncio -from functools import wraps import inspect +from functools import wraps from twisted.internet import defer, task from twisted.python import failure from scrapy.exceptions import IgnoreRequest -from scrapy.utils.asyncio import is_asyncio_reactor_installed +from scrapy.utils.reactor import is_asyncio_reactor_installed def defer_fail(_failure): diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index e4cf0196b..afef2c93f 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -1,17 +1,16 @@ # -*- coding: utf-8 -*- -import sys import logging +import sys import warnings from logging.config import dictConfig -from twisted.python.failure import Failure from twisted.python import log as twisted_log +from twisted.python.failure import Failure import scrapy -from scrapy.settings import Settings from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.asyncio import is_asyncio_reactor_installed +from scrapy.settings import Settings from scrapy.utils.versions import scrapy_components_versions @@ -149,8 +148,8 @@ def log_scrapy_info(settings): {'versions': ", ".join("%s %s" % (name, version) for name, version in scrapy_components_versions() if name != "Scrapy")}) - if is_asyncio_reactor_installed(): - logger.debug("Asyncio reactor is installed") + from twisted.internet import reactor + logger.debug("Using reactor: %s.%s", reactor.__module__, reactor.__class__.__name__) class StreamLogger(object): diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index b98fff6ec..80f52a4ef 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -1,4 +1,9 @@ -from twisted.internet import error +import asyncio +from contextlib import suppress + +from twisted.internet import asyncioreactor, error + +from scrapy.utils.misc import load_object def listen_tcp(portrange, host, factory): @@ -42,3 +47,31 @@ class CallLaterOnce(object): def __call__(self): self._call = None return self._func(*self._a, **self._kw) + + +def install_reactor(reactor_path): + reactor_class = load_object(reactor_path) + if reactor_class is asyncioreactor.AsyncioSelectorReactor: + with suppress(error.ReactorAlreadyInstalledError): + asyncioreactor.install(asyncio.get_event_loop()) + else: + *module, _ = reactor_path.split(".") + installer_path = module + ["install"] + installer = load_object(".".join(installer_path)) + with suppress(error.ReactorAlreadyInstalledError): + installer() + + +def verify_installed_reactor(reactor_path): + from twisted.internet import reactor + reactor_class = load_object(reactor_path) + if not isinstance(reactor, reactor_class): + msg = "The installed reactor ({}.{}) does not match the requested one ({})".format( + reactor.__module__, reactor.__class__.__name__, reactor_path + ) + raise Exception(msg) + + +def is_asyncio_reactor_installed(): + from twisted.internet import reactor + return isinstance(reactor, asyncioreactor.AsyncioSelectorReactor) diff --git a/tests/CrawlerProcess/asyncio_enabled_no_reactor.py b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py index db1b75931..d1e4a7bb5 100644 --- a/tests/CrawlerProcess/asyncio_enabled_no_reactor.py +++ b/tests/CrawlerProcess/asyncio_enabled_no_reactor.py @@ -10,8 +10,7 @@ class NoRequestsSpider(scrapy.Spider): process = CrawlerProcess(settings={ - 'ASYNCIO_REACTOR': True, + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", }) - process.crawl(NoRequestsSpider) process.start() diff --git a/tests/CrawlerProcess/asyncio_enabled_reactor.py b/tests/CrawlerProcess/asyncio_enabled_reactor.py index cec3c9c25..8568bd8b8 100644 --- a/tests/CrawlerProcess/asyncio_enabled_reactor.py +++ b/tests/CrawlerProcess/asyncio_enabled_reactor.py @@ -15,8 +15,7 @@ class NoRequestsSpider(scrapy.Spider): process = CrawlerProcess(settings={ - 'ASYNCIO_REACTOR': True, + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", }) - process.crawl(NoRequestsSpider) process.start() diff --git a/tests/CrawlerProcess/twisted_reactor_asyncio.py b/tests/CrawlerProcess/twisted_reactor_asyncio.py new file mode 100644 index 000000000..c6cbf949b --- /dev/null +++ b/tests/CrawlerProcess/twisted_reactor_asyncio.py @@ -0,0 +1,13 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class AsyncioReactorSpider(scrapy.Spider): + name = 'asyncio_reactor' + + +process = CrawlerProcess(settings={ + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", +}) +process.crawl(AsyncioReactorSpider) +process.start() diff --git a/tests/CrawlerProcess/twisted_reactor_poll.py b/tests/CrawlerProcess/twisted_reactor_poll.py new file mode 100644 index 000000000..27063260b --- /dev/null +++ b/tests/CrawlerProcess/twisted_reactor_poll.py @@ -0,0 +1,13 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class PollReactorSpider(scrapy.Spider): + name = 'poll_reactor' + + +process = CrawlerProcess(settings={ + "TWISTED_REACTOR": "twisted.internet.pollreactor.PollReactor", +}) +process.crawl(PollReactorSpider) +process.start() diff --git a/tests/CrawlerProcess/twisted_reactor_select.py b/tests/CrawlerProcess/twisted_reactor_select.py new file mode 100644 index 000000000..9af8ceb4d --- /dev/null +++ b/tests/CrawlerProcess/twisted_reactor_select.py @@ -0,0 +1,13 @@ +import scrapy +from scrapy.crawler import CrawlerProcess + + +class SelectReactorSpider(scrapy.Spider): + name = 'epoll_reactor' + + +process = CrawlerProcess(settings={ + "TWISTED_REACTOR": "twisted.internet.selectreactor.SelectReactor", +}) +process.crawl(SelectReactorSpider) +process.start() diff --git a/tests/test_commands.py b/tests/test_commands.py index 6024af71c..3612b70c9 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -296,12 +296,14 @@ class BadSpider(scrapy.Spider): self.assertIn("badspider.py", log) def test_asyncio_enabled_true(self): - log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_REACTOR=True']) - self.assertIn("DEBUG: Asyncio reactor is installed", log) + log = self.get_log(self.debug_log_spider, args=[ + '-s', 'TWISTED_REACTOR=twisted.internet.asyncioreactor.AsyncioSelectorReactor' + ]) + self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log) def test_asyncio_enabled_false(self): - log = self.get_log(self.debug_log_spider, args=['-s', 'ASYNCIO_REACTOR=False']) - self.assertNotIn("DEBUG: Asyncio reactor is installed", log) + log = self.get_log(self.debug_log_spider, args=[]) + self.assertNotIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log) class BenchCommandTest(CommandTest): diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 0ce0674de..f8fa26def 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -254,30 +254,38 @@ class CrawlerRunnerHasSpider(unittest.TestCase): def test_crawler_runner_asyncio_enabled_true(self): if self.reactor_pytest == 'asyncio': - runner = CrawlerRunner(settings={'ASYNCIO_REACTOR': True}) + runner = CrawlerRunner(settings={ + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", + }) else: - msg = "ASYNCIO_REACTOR is on but the Twisted asyncio reactor is not installed" + msg = r"The installed reactor \(.*?\) does not match the requested one \(.*?\)" with self.assertRaisesRegex(Exception, msg): - runner = CrawlerRunner(settings={'ASYNCIO_REACTOR': True}) + runner = CrawlerRunner(settings={ + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", + }) @defer.inlineCallbacks def test_crawler_process_asyncio_enabled_true(self): with LogCapture(level=logging.DEBUG) as log: if self.reactor_pytest == 'asyncio': - runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': True}) + runner = CrawlerProcess(settings={ + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", + }) yield runner.crawl(NoRequestsSpider) - self.assertIn("Asyncio reactor is installed", str(log)) + self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", str(log)) else: - msg = "ASYNCIO_REACTOR is on but the Twisted asyncio reactor is not installed" + msg = r"The installed reactor \(.*?\) does not match the requested one \(.*?\)" with self.assertRaisesRegex(Exception, msg): - runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': True}) + runner = CrawlerProcess(settings={ + "TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor", + }) @defer.inlineCallbacks def test_crawler_process_asyncio_enabled_false(self): - runner = CrawlerProcess(settings={'ASYNCIO_REACTOR': False}) + runner = CrawlerProcess(settings={"TWISTED_REACTOR": None}) with LogCapture(level=logging.DEBUG) as log: yield runner.crawl(NoRequestsSpider) - self.assertNotIn("Asyncio reactor is installed", str(log)) + self.assertNotIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", str(log)) class CrawlerProcessSubprocess(unittest.TestCase): @@ -294,17 +302,17 @@ class CrawlerProcessSubprocess(unittest.TestCase): def test_simple(self): log = self.run_script('simple.py') self.assertIn('Spider closed (finished)', log) - self.assertNotIn("DEBUG: Asyncio reactor is installed", log) + self.assertNotIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log) def test_asyncio_enabled_no_reactor(self): log = self.run_script('asyncio_enabled_no_reactor.py') self.assertIn('Spider closed (finished)', log) - self.assertIn("DEBUG: Asyncio reactor is installed", log) + self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log) def test_asyncio_enabled_reactor(self): log = self.run_script('asyncio_enabled_reactor.py') self.assertIn('Spider closed (finished)', log) - self.assertIn("DEBUG: Asyncio reactor is installed", log) + self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log) def test_ipv6_default_name_resolver(self): log = self.run_script('default_name_resolver.py') @@ -323,3 +331,18 @@ class CrawlerProcessSubprocess(unittest.TestCase): "'downloader/exception_type_count/twisted.internet.error.ConnectionRefusedError': 1," in log, "'downloader/exception_type_count/twisted.internet.error.ConnectError': 1," in log, ])) + + def test_reactor_select(self): + log = self.run_script("twisted_reactor_select.py") + self.assertIn("Spider closed (finished)", log) + self.assertIn("Using reactor: twisted.internet.selectreactor.SelectReactor", log) + + def test_reactor_poll(self): + log = self.run_script("twisted_reactor_poll.py") + self.assertIn("Spider closed (finished)", log) + self.assertIn("Using reactor: twisted.internet.pollreactor.PollReactor", log) + + def test_reactor_asyncio(self): + log = self.run_script("twisted_reactor_asyncio.py") + self.assertIn("Spider closed (finished)", log) + self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log) diff --git a/tests/test_utils_asyncio.py b/tests/test_utils_asyncio.py index 44acc24af..295323e4d 100644 --- a/tests/test_utils_asyncio.py +++ b/tests/test_utils_asyncio.py @@ -2,7 +2,7 @@ from unittest import TestCase from pytest import mark -from scrapy.utils.asyncio import is_asyncio_reactor_installed, install_asyncio_reactor +from scrapy.utils.reactor import is_asyncio_reactor_installed, install_reactor @mark.usefixtures('reactor_pytest') @@ -14,4 +14,4 @@ class AsyncioTest(TestCase): def test_install_asyncio_reactor(self): # this should do nothing - install_asyncio_reactor() + install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor") From 3263441fbcec8f46d363926d9106572cb0ecac5e Mon Sep 17 00:00:00 2001 From: Lane Shaw <lanethegreat@gmail.com> Date: Thu, 6 Feb 2020 16:14:40 -0500 Subject: [PATCH 192/313] Update RFPDupeFilter line separator for correct universal newlines mode usage (#4283) --- scrapy/dupefilters.py | 2 +- tests/test_dupefilters.py | 48 +++++++++++++++++++++++++++++++-------- 2 files changed, 40 insertions(+), 10 deletions(-) diff --git a/scrapy/dupefilters.py b/scrapy/dupefilters.py index ea6a4cfc3..a36c8304f 100644 --- a/scrapy/dupefilters.py +++ b/scrapy/dupefilters.py @@ -49,7 +49,7 @@ class RFPDupeFilter(BaseDupeFilter): return True self.fingerprints.add(fp) if self.file: - self.file.write(fp + os.linesep) + self.file.write(fp + '\n') def request_fingerprint(self, request): return request_fingerprint(request) diff --git a/tests/test_dupefilters.py b/tests/test_dupefilters.py index 0546558bc..88ce9627f 100644 --- a/tests/test_dupefilters.py +++ b/tests/test_dupefilters.py @@ -2,6 +2,8 @@ import hashlib import tempfile import unittest import shutil +import os +import sys from testfixtures import LogCapture from scrapy.dupefilters import RFPDupeFilter @@ -84,17 +86,21 @@ class RFPDupeFilterTest(unittest.TestCase): path = tempfile.mkdtemp() try: df = RFPDupeFilter(path) - df.open() - assert not df.request_seen(r1) - assert df.request_seen(r1) - df.close('finished') + try: + df.open() + assert not df.request_seen(r1) + assert df.request_seen(r1) + finally: + df.close('finished') df2 = RFPDupeFilter(path) - df2.open() - assert df2.request_seen(r1) - assert not df2.request_seen(r2) - assert df2.request_seen(r2) - df2.close('finished') + try: + df2.open() + assert df2.request_seen(r1) + assert not df2.request_seen(r2) + assert df2.request_seen(r2) + finally: + df2.close('finished') finally: shutil.rmtree(path) @@ -129,6 +135,30 @@ class RFPDupeFilterTest(unittest.TestCase): case_insensitive_dupefilter.close('finished') + def test_seenreq_newlines(self): + """ Checks against adding duplicate \r to + line endings on Windows platforms. """ + + r1 = Request('http://scrapytest.org/1') + + path = tempfile.mkdtemp() + try: + df = RFPDupeFilter(path) + df.open() + df.request_seen(r1) + df.close('finished') + + with open(os.path.join(path, 'requests.seen'), 'rb') as seen_file: + line = next(seen_file).decode() + assert not line.endswith('\r\r\n') + if sys.platform == 'win32': + assert line.endswith('\r\n') + else: + assert line.endswith('\n') + + finally: + shutil.rmtree(path) + def test_log(self): with LogCapture() as l: settings = {'DUPEFILTER_DEBUG': False, From 4f31c3ce017db2b4f69949a81efd56fee60ee32d Mon Sep 17 00:00:00 2001 From: Joy Bhalla <joybhalla9@gmail.com> Date: Fri, 7 Feb 2020 02:51:33 +0530 Subject: [PATCH 193/313] Document a backward incompatibility that may affect custom schedulers (#4274) --- docs/news.rst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/news.rst b/docs/news.rst index 6d0d4b4ee..e4b985c77 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -288,6 +288,13 @@ Backward-incompatible changes :class:`~scrapy.http.Request` objects instead of arbitrary Python data structures. +* An additional ``crawler`` parameter has been added to the ``__init__`` method + of the :class:`scrapy.core.scheduler.Scheduler` class. + Custom scheduler subclasses which don't accept arbitrary parameters in + their ``__init__`` method might break because of this change. + + For more information, refer to the documentation for the :setting:`SCHEDULER` setting. + See also :ref:`1.7-deprecation-removals` below. From 84b55b73646acca71461366ef98a1a501331f5d8 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Fri, 7 Feb 2020 11:07:35 +0500 Subject: [PATCH 194/313] Update docs/topics/signals.rst MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves <adrian@chaves.io> --- docs/topics/signals.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 47be6b603..60d9ce2bc 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -306,7 +306,7 @@ request_left_downloader The signal does not support returning deferreds from their handlers. - :param request: the request that reached downloader + :param request: the request that reached the downloader :type request: :class:`~scrapy.http.Request` object :param spider: the spider that yielded the request From 2f83f3e2cb3497e89d42533b8f20f8398a696f46 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Fri, 7 Feb 2020 11:07:43 +0500 Subject: [PATCH 195/313] Update docs/topics/signals.rst MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves <adrian@chaves.io> --- docs/topics/signals.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 60d9ce2bc..49475c1af 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -301,7 +301,7 @@ request_left_downloader .. signal:: request_left_downloader .. function:: request_left_downloader(request, spider) - Sent when a :class:`~scrapy.http.Request` leaves the downloader even in case of + Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of failure. The signal does not support returning deferreds from their handlers. From 8817b9e8e92f01147e7e44dd767165766093f408 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Fri, 7 Feb 2020 11:07:53 +0500 Subject: [PATCH 196/313] Update docs/topics/signals.rst MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves <adrian@chaves.io> --- docs/topics/signals.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 49475c1af..a7d60e9cb 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -296,7 +296,7 @@ request_reached_downloader :type spider: :class:`~scrapy.spiders.Spider` object request_left_downloader ---------------------------- +----------------------- .. signal:: request_left_downloader .. function:: request_left_downloader(request, spider) From 153b78e53f5c0f4630d09d560e111ca68f357905 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Fri, 7 Feb 2020 11:08:55 +0500 Subject: [PATCH 197/313] Update docs/topics/signals.rst Co-Authored-By: elacuesta <elacuesta@users.noreply.github.com> --- docs/topics/signals.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index a7d60e9cb..886d1b866 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -304,7 +304,7 @@ request_left_downloader Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of failure. - The signal does not support returning deferreds from their handlers. + This signal does not support returning deferreds from its handlers. :param request: the request that reached the downloader :type request: :class:`~scrapy.http.Request` object From 31f6c7112fe8efce2105983b8350b1dabdce7a1c Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Fri, 7 Feb 2020 17:14:52 +0500 Subject: [PATCH 198/313] Add a test for an async callbacks that returns requests. --- tests/spiders.py | 18 ++++++++++++++++++ tests/test_crawl.py | 14 ++++++++++++-- 2 files changed, 30 insertions(+), 2 deletions(-) diff --git a/tests/spiders.py b/tests/spiders.py index 3b1ee94b8..284c77829 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -117,6 +117,24 @@ class AsyncDefAsyncioReturnSpider(SimpleSpider): return [{'id': 1}, {'id': 2}] +class AsyncDefAsyncioReqsReturnSpider(SimpleSpider): + + name = 'asyncdef_asyncio_reqs_return' + + async def parse(self, response): + await asyncio.sleep(0.2) + req_id = response.meta.get('req_id', 0) + status = await get_from_asyncio_queue(response.status) + self.logger.info("Got response %d, req_id %d" % (status, req_id)) + if req_id > 0: + return + reqs = [] + for i in range(1, 3): + req = Request(self.start_urls[0], dont_filter=True, meta={'req_id': i}) + reqs.append(req) + return reqs + + class ItemSpider(FollowAllSpider): name = 'item' diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 85005eba4..b4b5bac1c 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -13,7 +13,8 @@ from scrapy.utils.python import to_unicode from tests.mockserver import MockServer from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider, SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback, - AsyncDefSpider, AsyncDefAsyncioSpider, AsyncDefAsyncioReturnSpider) + AsyncDefSpider, AsyncDefAsyncioSpider, AsyncDefAsyncioReturnSpider, + AsyncDefAsyncioReqsReturnSpider) class CrawlTestCase(TestCase): @@ -330,7 +331,7 @@ with multiples lines @mark.only_asyncio() @defer.inlineCallbacks - def test_async_def_asyncio_parse_list(self): + def test_async_def_asyncio_parse_items_list(self): items = [] def _on_item_scraped(item): @@ -343,3 +344,12 @@ with multiples lines self.assertIn("Got response 200", str(log)) self.assertIn({'id': 1}, items) self.assertIn({'id': 2}, items) + + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncio_parse_reqs_list(self): + crawler = self.runner.create_crawler(AsyncDefAsyncioReqsReturnSpider) + with LogCapture() as log: + yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + for req_id in range(3): + self.assertIn("Got response 200, req_id %d" % req_id, str(log)) From 7323780c97e69b560bf9a4bd7e6ccd60fb2b8f13 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 31 Dec 2019 16:15:41 +0500 Subject: [PATCH 199/313] Support yield in async def callbacks. --- conftest.py | 4 ++- scrapy/utils/py36.py | 10 ++++++++ scrapy/utils/spider.py | 8 ++++++ tests/py36/_test_crawl.py | 50 ++++++++++++++++++++++++++++++++++++++ tests/test_crawl.py | 51 +++++++++++++++++++++++++++++++++++++++ 5 files changed, 122 insertions(+), 1 deletion(-) create mode 100644 scrapy/utils/py36.py create mode 100644 tests/py36/_test_crawl.py diff --git a/conftest.py b/conftest.py index c0de09909..be5fbabf4 100644 --- a/conftest.py +++ b/conftest.py @@ -11,7 +11,9 @@ collect_ignore = [ # not a test, but looks like a test "scrapy/utils/testsite.py", # contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess - *_py_files("tests/CrawlerProcess") + *_py_files("tests/CrawlerProcess"), + # Py36-only parts of respective tests + *_py_files("tests/py36"), ] for line in open('tests/ignores.txt'): diff --git a/scrapy/utils/py36.py b/scrapy/utils/py36.py new file mode 100644 index 000000000..c8c24076e --- /dev/null +++ b/scrapy/utils/py36.py @@ -0,0 +1,10 @@ +""" +Helpers using Python 3.6+ syntax (ignore SyntaxError on import). +""" + + +async def collect_asyncgen(result): + results = [] + async for x in result: + results.append(x) + return results diff --git a/scrapy/utils/spider.py b/scrapy/utils/spider.py index 72775df5c..4e2a4d1bc 100644 --- a/scrapy/utils/spider.py +++ b/scrapy/utils/spider.py @@ -4,12 +4,20 @@ import inspect from scrapy.spiders import Spider from scrapy.utils.defer import deferred_from_coro from scrapy.utils.misc import arg_to_iter +try: + from scrapy.utils.py36 import collect_asyncgen +except SyntaxError: + collect_asyncgen = None logger = logging.getLogger(__name__) def iterate_spider_output(result): + if collect_asyncgen and hasattr(inspect, 'isasyncgen') and inspect.isasyncgen(result): + d = deferred_from_coro(collect_asyncgen(result)) + d.addCallback(iterate_spider_output) + return d return arg_to_iter(deferred_from_coro(result)) diff --git a/tests/py36/_test_crawl.py b/tests/py36/_test_crawl.py new file mode 100644 index 000000000..74c7daf53 --- /dev/null +++ b/tests/py36/_test_crawl.py @@ -0,0 +1,50 @@ +import asyncio + +from scrapy import Request +from tests.spiders import SimpleSpider + + +class AsyncDefAsyncioGenSpider(SimpleSpider): + + name = 'asyncdef_asyncio_gen' + + async def parse(self, response): + await asyncio.sleep(0.2) + yield {'foo': 42} + self.logger.info("Got response %d" % response.status) + + +class AsyncDefAsyncioGenLoopSpider(SimpleSpider): + + name = 'asyncdef_asyncio_gen_loop' + + async def parse(self, response): + for i in range(10): + await asyncio.sleep(0.1) + yield {'foo': i} + self.logger.info("Got response %d" % response.status) + + +class AsyncDefAsyncioGenComplexSpider(SimpleSpider): + + name = 'asyncdef_asyncio_gen_complex' + initial_reqs = 4 + following_reqs = 3 + depth = 2 + + def _get_req(self, index): + return Request(self.mockserver.url("/status?n=200&request=%d" % index), + meta={'index': index}) + + def start_requests(self): + for i in range(self.initial_reqs): + yield self._get_req(i) + + async def parse(self, response): + index = response.meta['index'] + yield {'index': index} + if index < 10 ** self.depth: + for new_index in range(10 * index, 10 * index + self.following_reqs): + yield self._get_req(new_index) + await asyncio.sleep(0.1) + yield {'index': index + 5} diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 85005eba4..856068465 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -1,5 +1,6 @@ import json import logging +import sys from pytest import mark from testfixtures import LogCapture @@ -343,3 +344,53 @@ with multiples lines self.assertIn("Got response 200", str(log)) self.assertIn({'id': 1}, items) self.assertIn({'id': 2}, items) + + @mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher") + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncgen_parse(self): + from tests.py36._test_crawl import AsyncDefAsyncioGenSpider + crawler = self.runner.create_crawler(AsyncDefAsyncioGenSpider) + with LogCapture() as log: + yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + self.assertIn("Got response 200", str(log)) + itemcount = crawler.stats.get_value('item_scraped_count') + self.assertEqual(itemcount, 1) + + @mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher") + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncgen_parse_loop(self): + items = [] + + def _on_item_scraped(item): + items.append(item) + + from tests.py36._test_crawl import AsyncDefAsyncioGenLoopSpider + crawler = self.runner.create_crawler(AsyncDefAsyncioGenLoopSpider) + crawler.signals.connect(_on_item_scraped, signals.item_scraped) + with LogCapture() as log: + yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) + self.assertIn("Got response 200", str(log)) + itemcount = crawler.stats.get_value('item_scraped_count') + self.assertEqual(itemcount, 10) + for i in range(10): + self.assertIn({'foo': i}, items) + + @mark.skipif(sys.version_info < (3, 6), reason="Async generators require Python 3.6 or higher") + @mark.only_asyncio() + @defer.inlineCallbacks + def test_async_def_asyncgen_parse_complex(self): + items = [] + + def _on_item_scraped(item): + items.append(item) + + from tests.py36._test_crawl import AsyncDefAsyncioGenComplexSpider + crawler = self.runner.create_crawler(AsyncDefAsyncioGenComplexSpider) + crawler.signals.connect(_on_item_scraped, signals.item_scraped) + yield crawler.crawl(mockserver=self.mockserver) + itemcount = crawler.stats.get_value('item_scraped_count') + self.assertEqual(itemcount, 80) + for i in [0, 3, 21, 22, 207, 311]: # some random items + self.assertIn({'index': i}, items) From 59653ebac609dc11c9d3b29624972d2aaddd5541 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Fri, 7 Feb 2020 21:07:57 +0100 Subject: [PATCH 200/313] Update installation instructions regarding Python 3 and virtual environments --- docs/intro/install.rst | 32 ++++++-------------------------- 1 file changed, 6 insertions(+), 26 deletions(-) diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 178be723c..49968437c 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -81,35 +81,18 @@ Python packages can be installed either globally (a.k.a system wide), or in user-space. We do not recommend installing Scrapy system wide. Instead, we recommend that you install Scrapy within a so-called -"virtual environment" (`virtualenv`_). -Virtualenvs allow you to not conflict with already-installed Python +"virtual environment" (:mod:`venv`). +Virtual environments allow you to not conflict with already-installed Python system packages (which could break some of your system tools and scripts), and still install packages normally with ``pip`` (without ``sudo`` and the likes). -To get started with virtual environments, see `virtualenv installation instructions`_. -To install it globally (having it globally installed actually helps here), -it should be a matter of running:: +See :ref:`tut-venv` on how to create your virtual environment. - $ [sudo] pip install virtualenv - -Check this `user guide`_ on how to create your virtualenv. - -.. note:: - If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs. - -Once you have created a virtualenv, you can install Scrapy inside it with ``pip``, +Once you have created a virtual environment, you can install Scrapy inside it with ``pip``, just like any other Python package. (See :ref:`platform-specific guides <intro-install-platform-notes>` below for non-Python dependencies that you may need to install beforehand). -Python virtualenvs can be created to use Python 2 by default, or Python 3 by default. As Scrapy -only supports Python 3, make sure you created a Python 3 virtualenv. - -.. _virtualenv: https://virtualenv.pypa.io -.. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/ -.. _virtualenvwrapper: https://virtualenvwrapper.readthedocs.io/en/latest/install.html -.. _user guide: https://virtualenv.pypa.io/en/stable/userguide/ - .. _intro-install-platform-notes: @@ -205,15 +188,12 @@ solutions: brew update; brew upgrade python -* *(Optional)* Install Scrapy inside an isolated python environment. +* *(Optional)* :ref:`Install Scrapy inside a Python virtual environment + <intro-using-virtualenv>`. This method is a workaround for the above OS X issue, but it's an overall good practice for managing dependencies and can complement the first method. - `virtualenv`_ is a tool you can use to create virtual environments in python. - We recommended reading a tutorial like - http://docs.python-guide.org/en/latest/dev/virtualenvs/ to get started. - After any of these workarounds you should be able to install Scrapy:: pip install Scrapy From 35723d76c0c07575309810e776305e9ea22fc18d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Fri, 7 Feb 2020 22:59:53 +0100 Subject: [PATCH 201/313] Use canonicalize_url in link extraction --- scrapy/linkextractors/lxmlhtml.py | 4 ++-- tests/test_linkextractors.py | 5 +---- 2 files changed, 3 insertions(+), 6 deletions(-) diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index fdfa92370..da525d52e 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -9,7 +9,7 @@ from w3lib.url import canonicalize_url from scrapy.link import Link from scrapy.utils.misc import arg_to_iter, rel_has_nofollow -from scrapy.utils.python import unique as unique_list, to_unicode +from scrapy.utils.python import unique as unique_list from scrapy.utils.response import get_base_url from scrapy.linkextractors import FilteringLinkExtractor @@ -66,7 +66,7 @@ class LxmlParserLinkExtractor(object): url = self.process_attr(attr_val) if url is None: continue - url = to_unicode(url, encoding=response_encoding) + url = canonicalize_url(url, encoding=response_encoding) # to fix relative links after process_value url = urljoin(response_url, url) link = Link(url, _collect_string_content(el) or u'', diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index 38fb8fb4a..e9d6c0abe 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -2,8 +2,6 @@ import re import unittest from warnings import catch_warnings -import pytest - from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import HtmlResponse, XmlResponse from scrapy.link import Link @@ -214,7 +212,7 @@ class Base: response = HtmlResponse("http://example.org/somepage/index.html", body=html, encoding='iso8859-15') links = self.extractor_cls(restrict_xpaths='//p').extract_links(response) self.assertEqual(links, - [Link(url='http://example.org/%E2%99%A5/you?c=%E2%82%AC', text=u'text')]) + [Link(url='http://example.org/%E2%99%A5/you?c=%A4', text=u'text')]) def test_restrict_xpaths_concat_in_handle_data(self): """html entities cause SGMLParser to call handle_data hook twice""" @@ -506,7 +504,6 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase): Link(url='http://example.org/item2.html', text=u'Pic of a dog', nofollow=False), ]) - @pytest.mark.xfail def test_restrict_xpaths_with_html_entities(self): super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities() From 13ba9bc629cb0a77ebaca36a10a0a4984d7cce68 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 10 Feb 2020 12:29:39 -0300 Subject: [PATCH 202/313] Note about Response.ip_address --- docs/topics/request-response.rst | 2 ++ 1 file changed, 2 insertions(+) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 17eb63064..89e570028 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -685,6 +685,8 @@ Response objects .. attribute:: Response.ip_address The IP address of the server from which the Response originated. + This attribute is currently only populated by the HTTP 1.1 download + handler, i.e. for ``http(s)`` responses. .. method:: Response.copy() From 4626e90df8ba4a945bb9cd6be47a915788e76f23 Mon Sep 17 00:00:00 2001 From: Abhishek Pratap Singh <35230163+Prime-5@users.noreply.github.com> Date: Mon, 10 Feb 2020 18:48:31 +0000 Subject: [PATCH 203/313] Allow updating flags in follow and follow_all (#4279) --- scrapy/http/response/__init__.py | 7 +++++-- scrapy/http/response/text.py | 6 ++++-- tests/test_http_response.py | 31 +++++++++++++++++++++++++++++++ 3 files changed, 40 insertions(+), 4 deletions(-) diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index f92d0901c..027fbac6f 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -107,7 +107,7 @@ class Response(object_ref): def follow(self, url, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding='utf-8', priority=0, - dont_filter=False, errback=None, cb_kwargs=None): + dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Request """ Return a :class:`~.Request` instance to follow a link ``url``. @@ -124,6 +124,7 @@ class Response(object_ref): elif url is None: raise ValueError("url can't be None") url = self.urljoin(url) + return Request( url=url, callback=callback, @@ -137,11 +138,12 @@ class Response(object_ref): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) def follow_all(self, urls, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding='utf-8', priority=0, - dont_filter=False, errback=None, cb_kwargs=None): + dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Generator[Request, None, None] """ Return an iterable of :class:`~.Request` instances to follow all links @@ -169,6 +171,7 @@ class Response(object_ref): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) for url in urls ) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 09049c157..33a485328 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -121,7 +121,7 @@ class TextResponse(Response): def follow(self, url, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding=None, priority=0, - dont_filter=False, errback=None, cb_kwargs=None): + dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Request """ Return a :class:`~.Request` instance to follow a link ``url``. @@ -157,11 +157,12 @@ class TextResponse(Response): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) def follow_all(self, urls=None, callback=None, method='GET', headers=None, body=None, cookies=None, meta=None, encoding=None, priority=0, - dont_filter=False, errback=None, cb_kwargs=None, + dont_filter=False, errback=None, cb_kwargs=None, flags=None, css=None, xpath=None): # type: (...) -> Generator[Request, None, None] """ @@ -214,6 +215,7 @@ class TextResponse(Response): dont_filter=dont_filter, errback=errback, cb_kwargs=cb_kwargs, + flags=flags, ) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 4c1b2afc3..ff487cfa3 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -166,6 +166,10 @@ class BaseResponseTest(unittest.TestCase): def test_follow_whitespace_link(self): self._assert_followed_url(Link('http://example.com/foo '), 'http://example.com/foo%20') + def test_follow_flags(self): + res = self.response_class('http://example.com/') + fol = res.follow('http://example.com/', flags=['cached', 'allowed']) + self.assertEqual(fol.flags, ['cached', 'allowed']) # Response.follow_all @@ -232,6 +236,17 @@ class BaseResponseTest(unittest.TestCase): expected = [u.replace(' ', '%20') for u in absolute] self._assert_followed_all_urls(links, expected) + def test_follow_all_flags(self): + re = self.response_class('http://www.example.com/') + urls = [ + 'http://www.example.com/', + 'http://www.example.com/2', + 'http://www.example.com/foo', + ] + fol = re.follow_all(urls, flags=['cached', 'allowed']) + for req in fol: + self.assertEqual(req.flags, ['cached', 'allowed']) + def _assert_followed_url(self, follow_obj, target_url, response=None): if response is None: response = self._links_response() @@ -562,6 +577,22 @@ class TextResponseTest(BaseResponseTest): ) self.assertEqual(req.encoding, 'cp1251') + def test_follow_flags(self): + res = self.response_class('http://example.com/') + fol = res.follow('http://example.com/', flags=['cached', 'allowed']) + self.assertEqual(fol.flags, ['cached', 'allowed']) + + def test_follow_all_flags(self): + re = self.response_class('http://www.example.com/') + urls = [ + 'http://www.example.com/', + 'http://www.example.com/2', + 'http://www.example.com/foo', + ] + fol = re.follow_all(urls, flags=['cached', 'allowed']) + for req in fol: + self.assertEqual(req.flags, ['cached', 'allowed']) + def test_follow_all_css(self): expected = [ 'http://example.com/sample3.html', From 037ae5b22e6d6600dc537ee5073652ce74e5f47b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Mon, 10 Feb 2020 19:54:47 +0100 Subject: [PATCH 204/313] =?UTF-8?q?Explicitly=20indicate=20None=20as=20ip?= =?UTF-8?q?=5Faddress=E2=80=99s=20default=20value?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/topics/request-response.rst | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 89e570028..8f2504a33 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -685,8 +685,10 @@ Response objects .. attribute:: Response.ip_address The IP address of the server from which the Response originated. + This attribute is currently only populated by the HTTP 1.1 download - handler, i.e. for ``http(s)`` responses. + handler, i.e. for ``http(s)`` responses. For other handlers, + :attr:`ip_address` is always ``None``. .. method:: Response.copy() From 36dcf901849014d7db00a0294ed86c6cc79b5cc6 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 11 Feb 2020 00:57:58 +0500 Subject: [PATCH 205/313] Also test non-default async callbacks. --- tests/py36/_test_crawl.py | 13 ++++++++++--- tests/test_crawl.py | 7 +++++-- 2 files changed, 15 insertions(+), 5 deletions(-) diff --git a/tests/py36/_test_crawl.py b/tests/py36/_test_crawl.py index 74c7daf53..162a53760 100644 --- a/tests/py36/_test_crawl.py +++ b/tests/py36/_test_crawl.py @@ -32,12 +32,14 @@ class AsyncDefAsyncioGenComplexSpider(SimpleSpider): following_reqs = 3 depth = 2 - def _get_req(self, index): + def _get_req(self, index, cb=None): return Request(self.mockserver.url("/status?n=200&request=%d" % index), - meta={'index': index}) + meta={'index': index}, + dont_filter=True, + callback=cb) def start_requests(self): - for i in range(self.initial_reqs): + for i in range(1, self.initial_reqs + 1): yield self._get_req(i) async def parse(self, response): @@ -46,5 +48,10 @@ class AsyncDefAsyncioGenComplexSpider(SimpleSpider): if index < 10 ** self.depth: for new_index in range(10 * index, 10 * index + self.following_reqs): yield self._get_req(new_index) + yield self._get_req(index, cb=self.parse2) await asyncio.sleep(0.1) yield {'index': index + 5} + + async def parse2(self, response): + await asyncio.sleep(0.1) + yield {'index2': response.meta['index']} diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 626000147..64819acb6 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -392,9 +392,12 @@ with multiples lines crawler.signals.connect(_on_item_scraped, signals.item_scraped) yield crawler.crawl(mockserver=self.mockserver) itemcount = crawler.stats.get_value('item_scraped_count') - self.assertEqual(itemcount, 80) - for i in [0, 3, 21, 22, 207, 311]: # some random items + self.assertEqual(itemcount, 156) + # some random items + for i in [1, 4, 21, 22, 207, 311]: self.assertIn({'index': i}, items) + for i in [10, 30, 122]: + self.assertIn({'index2': i}, items) @mark.only_asyncio() @defer.inlineCallbacks From 1f0f52cbf7bdc9f11f7b83c482ad52ad7ad32ba0 Mon Sep 17 00:00:00 2001 From: Andrey Rakhmatullin <wrar@wrar.name> Date: Tue, 11 Feb 2020 01:05:45 +0500 Subject: [PATCH 206/313] Improve async signal tests. --- tests/test_signals.py | 15 ++++++++++----- 1 file changed, 10 insertions(+), 5 deletions(-) diff --git a/tests/test_signals.py b/tests/test_signals.py index 001e798e5..d6ae526be 100644 --- a/tests/test_signals.py +++ b/tests/test_signals.py @@ -1,8 +1,9 @@ +from pytest import mark from twisted.internet import defer from twisted.trial import unittest from scrapy import signals, Request, Spider -from scrapy.utils.test import get_crawler +from scrapy.utils.test import get_crawler, get_from_asyncio_queue from tests.mockserver import MockServer @@ -11,12 +12,12 @@ class ItemSpider(Spider): name = 'itemspider' def start_requests(self): - for _ in range(10): - yield Request(self.mockserver.url('/status?n=200'), - dont_filter=True) + for index in range(10): + yield Request(self.mockserver.url('/status?n=200&id=%d' % index), + meta={'index': index}) def parse(self, response): - return {'field': 42} + return {'index': response.meta['index']} class AsyncSignalTestCase(unittest.TestCase): @@ -29,11 +30,15 @@ class AsyncSignalTestCase(unittest.TestCase): self.mockserver.__exit__(None, None, None) async def _on_item_scraped(self, item): + item = await get_from_asyncio_queue(item) self.items.append(item) + @mark.only_asyncio() @defer.inlineCallbacks def test_simple_pipeline(self): crawler = get_crawler(ItemSpider) crawler.signals.connect(self._on_item_scraped, signals.item_scraped) yield crawler.crawl(mockserver=self.mockserver) self.assertEqual(len(self.items), 10) + for index in range(10): + self.assertIn({'index': index}, self.items) From 61e74bac765de0f786d2125e876e4d7934f1722b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Mon, 10 Feb 2020 21:57:21 +0100 Subject: [PATCH 207/313] Extract links with safe_url_string canonicalize_url changes links in undesirable ways. --- scrapy/linkextractors/lxmlhtml.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index da525d52e..f5ef56ea4 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -5,7 +5,7 @@ from urllib.parse import urljoin import lxml.etree as etree from w3lib.html import strip_html5_whitespace -from w3lib.url import canonicalize_url +from w3lib.url import canonicalize_url, safe_url_string from scrapy.link import Link from scrapy.utils.misc import arg_to_iter, rel_has_nofollow @@ -66,7 +66,7 @@ class LxmlParserLinkExtractor(object): url = self.process_attr(attr_val) if url is None: continue - url = canonicalize_url(url, encoding=response_encoding) + url = safe_url_string(url, encoding=response_encoding) # to fix relative links after process_value url = urljoin(response_url, url) link = Link(url, _collect_string_content(el) or u'', From 2d6d4fb2335ef24b1efca67dcedf5d264a642e0f Mon Sep 17 00:00:00 2001 From: Drew Seibert <drewjbert@gmail.com> Date: Tue, 11 Feb 2020 03:35:23 -0600 Subject: [PATCH 208/313] Deprecate overriding settings with SCRAPY-prefixed environment variables (#4300) --- scrapy/utils/project.py | 4 +--- 1 file changed, 1 insertion(+), 3 deletions(-) diff --git a/scrapy/utils/project.py b/scrapy/utils/project.py index f28c2eaa1..d9a03ff63 100644 --- a/scrapy/utils/project.py +++ b/scrapy/utils/project.py @@ -68,7 +68,6 @@ def get_project_settings(): if settings_module_path: settings.setmodule(settings_module_path, priority='project') - # XXX: remove this hack pickled_settings = os.environ.get("SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE") if pickled_settings: warnings.warn("Use of environment variable " @@ -76,10 +75,9 @@ def get_project_settings(): "is deprecated.", ScrapyDeprecationWarning) settings.setdict(pickle.loads(pickled_settings), priority='project') - # XXX: deprecate and remove this functionality env_overrides = {k[7:]: v for k, v in os.environ.items() if k.startswith('SCRAPY_')} if env_overrides: + warnings.warn("Use of 'SCRAPY_'-prefixed environment variables to override settings is deprecated.", ScrapyDeprecationWarning) settings.setdict(env_overrides, priority='project') - return settings From b4958358e89b6611f7dd852684ba6d599831fe27 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Wed, 12 Feb 2020 19:00:04 +0100 Subject: [PATCH 209/313] Update tests to account for link extractors escaping spaces --- tests/test_linkextractors.py | 11 ++--------- 1 file changed, 2 insertions(+), 9 deletions(-) diff --git a/tests/test_linkextractors.py b/tests/test_linkextractors.py index e9d6c0abe..53968e60e 100644 --- a/tests/test_linkextractors.py +++ b/tests/test_linkextractors.py @@ -14,7 +14,6 @@ from tests import get_testdata class Base: class LinkExtractorTestCase(unittest.TestCase): extractor_cls = None - escapes_whitespace = False def setUp(self): body = get_testdata('link_extractor', 'linkextractor.html') @@ -28,10 +27,7 @@ class Base: def test_extract_all_links(self): lx = self.extractor_cls() - if self.escapes_whitespace: - page4_url = 'http://example.com/page%204.html' - else: - page4_url = 'http://example.com/page 4.html' + page4_url = 'http://example.com/page%204.html' self.assertEqual([link for link in lx.extract_links(self.response)], [ Link(url='http://example.com/sample1.html', text=u''), @@ -308,10 +304,7 @@ class Base: def test_attrs(self): lx = self.extractor_cls(attrs="href") - if self.escapes_whitespace: - page4_url = 'http://example.com/page%204.html' - else: - page4_url = 'http://example.com/page 4.html' + page4_url = 'http://example.com/page%204.html' self.assertEqual(lx.extract_links(self.response), [ Link(url='http://example.com/sample1.html', text=u''), From df937d8280fe0781f6cf1715a3a0cd28c6e94eae Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 13 Feb 2020 22:33:36 +0100 Subject: [PATCH 210/313] Implement Response.cb_kwargs --- docs/topics/request-response.rst | 12 ++++++++++++ scrapy/http/response/__init__.py | 10 ++++++++++ 2 files changed, 22 insertions(+) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 4cf367d96..05b7bb5c7 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -672,6 +672,18 @@ Response objects .. seealso:: :attr:`Request.meta` attribute + .. attribute:: Response.cb_kwargs + + A shortcut to the :attr:`Request.cb_kwargs` attribute of the + :attr:`Response.request` object (ie. ``self.request.cb_kwargs``). + + Unlike the :attr:`Response.request` attribute, the + :attr:`Response.cb_kwargs` attribute is propagated along redirects and + retries, so you will get the original :attr:`Request.cb_kwargs` sent + from your spider. + + .. seealso:: :attr:`Request.cb_kwargs` attribute + .. attribute:: Response.flags A list that contains flags for this response. Flags are labels used for diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index 64e9c6c20..ee9720d52 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -24,6 +24,16 @@ class Response(object_ref): self.request = request self.flags = [] if flags is None else list(flags) + @property + def cb_kwargs(self): + try: + return self.request.cb_kwargs + except AttributeError: + raise AttributeError( + "Response.cb_kwargs not available, this response " + "is not tied to any request" + ) + @property def meta(self): try: From 5ff9eb90ea9d533d3f960db75071f0fe638503ab Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 13 Feb 2020 22:36:18 +0100 Subject: [PATCH 211/313] Add a test for the copy of cb_kwargs from Request to Response --- tests/test_http_response.py | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 960ecea3e..39f5fe750 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -72,6 +72,12 @@ class BaseResponseTest(unittest.TestCase): r1 = self.response_class("http://www.example.com", body=b"Some body", request=req) assert r1.meta is req.meta + def test_copy_cb_kwargs(self): + req = Request("http://www.example.com") + req.cb_kwargs['foo'] = 'bar' + r1 = self.response_class("http://www.example.com", body=b"Some body", request=req) + assert r1.cb_kwargs is req.cb_kwargs + def test_copy_inherited_classes(self): """Test Response children copies preserve their class""" From 43b43654a1dacae7b63fc067dc69929c262d9a15 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 13 Feb 2020 22:39:58 +0100 Subject: [PATCH 212/313] Add tests for meta and cb_kwargs not being available --- tests/test_http_response.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 39f5fe750..5a19f9d54 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -78,6 +78,16 @@ class BaseResponseTest(unittest.TestCase): r1 = self.response_class("http://www.example.com", body=b"Some body", request=req) assert r1.cb_kwargs is req.cb_kwargs + def test_unavailable_meta(self): + r1 = self.response_class("http://www.example.com", body=b"Some body") + with self.assertRaisesRegex(AttributeError, r'Response\.meta not available'): + r1.meta + + def test_unavailable_cb_kwargs(self): + r1 = self.response_class("http://www.example.com", body=b"Some body") + with self.assertRaisesRegex(AttributeError, r'Response\.cb_kwargs not available'): + r1.cb_kwargs + def test_copy_inherited_classes(self): """Test Response children copies preserve their class""" From 5ae3e1678fa99b3c44cb8981079df51ec34b860f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Fri, 14 Feb 2020 22:30:36 +0100 Subject: [PATCH 213/313] =?UTF-8?q?ie.=20=E2=86=92=20i.e.?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: elacuesta <elacuesta@users.noreply.github.com> --- docs/topics/request-response.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 05b7bb5c7..260fe3caf 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -675,7 +675,7 @@ Response objects .. attribute:: Response.cb_kwargs A shortcut to the :attr:`Request.cb_kwargs` attribute of the - :attr:`Response.request` object (ie. ``self.request.cb_kwargs``). + :attr:`Response.request` object (i.e. ``self.request.cb_kwargs``). Unlike the :attr:`Response.request` attribute, the :attr:`Response.cb_kwargs` attribute is propagated along redirects and From a04dd13cd08f1ff392a8bbe284fa0c8fe8924b57 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Fri, 14 Feb 2020 22:31:30 +0100 Subject: [PATCH 214/313] =?UTF-8?q?ie.=20=E2=86=92=20i.e.?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/topics/request-response.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 260fe3caf..d6c7cbec9 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -664,7 +664,7 @@ Response objects .. attribute:: Response.meta A shortcut to the :attr:`Request.meta` attribute of the - :attr:`Response.request` object (ie. ``self.request.meta``). + :attr:`Response.request` object (i.e. ``self.request.meta``). Unlike the :attr:`Response.request` attribute, the :attr:`Response.meta` attribute is propagated along redirects and retries, so you will get @@ -770,7 +770,7 @@ TextResponse objects 1. the encoding passed in the ``__init__`` method ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this - encoding is not valid (ie. unknown), it is ignored and the next + encoding is not valid (i.e. unknown), it is ignored and the next resolution mechanism is tried. 3. the encoding declared in the response body. The TextResponse class From 182445f9d96130b1041ece8c4b2a9e9891107c73 Mon Sep 17 00:00:00 2001 From: Akshay Sharma <42249933+AKSHAYSHARMAJS@users.noreply.github.com> Date: Tue, 18 Feb 2020 22:28:31 +0530 Subject: [PATCH 215/313] =?UTF-8?q?Fix=20a=20spelling=20error:=20ie.=20?= =?UTF-8?q?=E2=86=92=20i.e.=20(#4338)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- docs/intro/tutorial.rst | 2 +- docs/topics/downloader-middleware.rst | 4 ++-- docs/topics/extensions.rst | 6 +++--- docs/topics/feed-exports.rst | 2 +- docs/topics/jobs.rst | 2 +- docs/topics/link-extractors.rst | 2 +- docs/topics/request-response.rst | 4 ++-- docs/topics/selectors.rst | 4 ++-- docs/topics/settings.rst | 6 +++--- docs/topics/signals.rst | 4 ++-- scrapy/shell.py | 2 +- scrapy/utils/misc.py | 4 ++-- scrapy/utils/request.py | 2 +- scrapy/utils/spider.py | 2 +- sep/sep-003.rst | 4 ++-- sep/sep-013.rst | 2 +- sep/sep-021.rst | 2 +- 17 files changed, 27 insertions(+), 27 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index ee10048b5..798fe4a7a 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -212,7 +212,7 @@ using the :ref:`Scrapy shell <topics-shell>`. Run:: .. note:: Remember to always enclose urls in quotes when running Scrapy shell from - command-line, otherwise urls containing arguments (ie. ``&`` character) + command-line, otherwise urls containing arguments (i.e. ``&`` character) will not work. On Windows, use double quotes instead:: diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 3ec6e0c17..a83cedcfd 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -259,8 +259,8 @@ COOKIES_DEBUG Default: ``False`` -If enabled, Scrapy will log all cookies sent in requests (ie. ``Cookie`` -header) and all cookies received in responses (ie. ``Set-Cookie`` header). +If enabled, Scrapy will log all cookies sent in requests (i.e. ``Cookie`` +header) and all cookies received in responses (i.e. ``Set-Cookie`` header). Here's an example of a log with :setting:`COOKIES_DEBUG` enabled:: diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index 0a7455ec9..dc057f6b6 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -63,7 +63,7 @@ but disabled unless the :setting:`HTTPCACHE_ENABLED` setting is set. Disabling an extension ====================== -In order to disable an extension that comes enabled by default (ie. those +In order to disable an extension that comes enabled by default (i.e. those included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to ``None``. For example:: @@ -345,7 +345,7 @@ signal is received. The information dumped is the following: After the stack trace and engine status is dumped, the Scrapy process continues running normally. -This extension only works on POSIX-compliant platforms (ie. not Windows), +This extension only works on POSIX-compliant platforms (i.e. not Windows), because the `SIGQUIT`_ and `SIGUSR2`_ signals are not available on Windows. There are at least two ways to send Scrapy the `SIGQUIT`_ signal: @@ -370,7 +370,7 @@ running normally. For more info see `Debugging in Python`_. -This extension only works on POSIX-compliant platforms (ie. not Windows). +This extension only works on POSIX-compliant platforms (i.e. not Windows). .. _Python debugger: https://docs.python.org/2/library/pdb.html .. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/ diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index 7481b1a99..1d94807a4 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -301,7 +301,7 @@ FEED_STORE_EMPTY Default: ``False`` -Whether to export empty feeds (ie. feeds with no items). +Whether to export empty feeds (i.e. feeds with no items). .. setting:: FEED_STORAGES diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst index 8816a028c..c34ba336b 100644 --- a/docs/topics/jobs.rst +++ b/docs/topics/jobs.rst @@ -22,7 +22,7 @@ Job directory To enable persistence support you just need to define a *job directory* through the ``JOBDIR`` setting. This directory will be for storing all required data to -keep the state of a single job (ie. a spider run). It's important to note that +keep the state of a single job (i.e. a spider run). It's important to note that this directory must not be shared by different spiders, or even different jobs/runs of the same spider, as it's meant to be used for storing the state of a *single* job. diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index 2119cb8f8..8c8019438 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -49,7 +49,7 @@ LxmlLinkExtractor :type allow: a regular expression (or list of) :param deny: a single regular expression (or list of regular expressions) - that the (absolute) urls must match in order to be excluded (ie. not + that the (absolute) urls must match in order to be excluded (i.e. not extracted). It has precedence over the ``allow`` parameter. If not given (or empty) it won't exclude any links. :type deny: a regular expression (or list of) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 8997a7f19..34cc41a02 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -664,7 +664,7 @@ Response objects .. attribute:: Response.meta A shortcut to the :attr:`Request.meta` attribute of the - :attr:`Response.request` object (ie. ``self.request.meta``). + :attr:`Response.request` object (i.e. ``self.request.meta``). Unlike the :attr:`Response.request` attribute, the :attr:`Response.meta` attribute is propagated along redirects and retries, so you will get @@ -760,7 +760,7 @@ TextResponse objects 1. the encoding passed in the ``__init__`` method ``encoding`` argument 2. the encoding declared in the Content-Type HTTP header. If this - encoding is not valid (ie. unknown), it is ignored and the next + encoding is not valid (i.e. unknown), it is ignored and the next resolution mechanism is tried. 3. the encoding declared in the response body. The TextResponse class diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 8ec758b0e..c3d431e2a 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -986,7 +986,7 @@ a :class:`~scrapy.http.HtmlResponse` object like this:: sel = Selector(html_response) 1. Select all ``<h1>`` elements from an HTML response body, returning a list of - :class:`Selector` objects (ie. a :class:`SelectorList` object):: + :class:`Selector` objects (i.e. a :class:`SelectorList` object):: sel.xpath("//h1") @@ -1013,7 +1013,7 @@ instantiated with an :class:`~scrapy.http.XmlResponse` object:: sel = Selector(xml_response) 1. Select all ``<product>`` elements from an XML response body, returning a list - of :class:`Selector` objects (ie. a :class:`SelectorList` object):: + of :class:`Selector` objects (i.e. a :class:`SelectorList` object):: sel.xpath("//product") diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index fa63a5807..5394147da 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -248,7 +248,7 @@ CONCURRENT_REQUESTS Default: ``16`` -The maximum number of concurrent (ie. simultaneous) requests that will be +The maximum number of concurrent (i.e. simultaneous) requests that will be performed by the Scrapy downloader. .. setting:: CONCURRENT_REQUESTS_PER_DOMAIN @@ -258,7 +258,7 @@ CONCURRENT_REQUESTS_PER_DOMAIN Default: ``8`` -The maximum number of concurrent (ie. simultaneous) requests that will be +The maximum number of concurrent (i.e. simultaneous) requests that will be performed to any single domain. See also: :ref:`topics-autothrottle` and its @@ -272,7 +272,7 @@ CONCURRENT_REQUESTS_PER_IP Default: ``0`` -The maximum number of concurrent (ie. simultaneous) requests that will be +The maximum number of concurrent (i.e. simultaneous) requests that will be performed to any single IP. If non-zero, the :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` setting is ignored, and this one is used instead. In other words, concurrency limits will be applied per IP, not diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 886d1b866..d3cfb0307 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -141,7 +141,7 @@ item_error .. signal:: item_error .. function:: item_error(item, response, spider, failure) - Sent when a :ref:`topics-item-pipeline` generates an error (ie. raises + Sent when a :ref:`topics-item-pipeline` generates an error (i.e. raises an exception), except :exc:`~scrapy.exceptions.DropItem` exception. This signal supports returning deferreds from their handlers. @@ -232,7 +232,7 @@ spider_error .. signal:: spider_error .. function:: spider_error(failure, response, spider) - Sent when a spider callback generates an error (ie. raises an exception). + Sent when a spider callback generates an error (i.e. raises an exception). This signal does not support returning deferreds from their handlers. diff --git a/scrapy/shell.py b/scrapy/shell.py index a23b04df9..1d5341973 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -173,7 +173,7 @@ def _request_deferred(request): This returns a Deferred whose first pair of callbacks are the request callback and errback. The Deferred also triggers when the request - callback/errback is executed (ie. when the request is downloaded) + callback/errback is executed (i.e. when the request is downloaded) WARNING: Do not call request.replace() until after the deferred is called. """ diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index cb0ee5af3..a3e55d6ea 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -37,8 +37,8 @@ def arg_to_iter(arg): def load_object(path): """Load an object given its absolute object path, and return it. - object can be a class, function, variable or an instance. - path ie: 'scrapy.downloadermiddlewares.redirect.RedirectMiddleware' + object can be the import path of a class, function, variable or an + instance, e.g. 'scrapy.downloadermiddlewares.redirect.RedirectMiddleware' """ try: diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index 356753ab5..b8c140a7e 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -28,7 +28,7 @@ def request_fingerprint(request, include_headers=None, keep_fragments=False): http://www.example.com/query?cat=222&id=111 Even though those are two different URLs both point to the same resource - and are equivalent (ie. they should return the same response). + and are equivalent (i.e. they should return the same response). Another example are cookies used to store session ids. Suppose the following page is only accessible to authenticated users: diff --git a/scrapy/utils/spider.py b/scrapy/utils/spider.py index 72775df5c..e4a2d1ac2 100644 --- a/scrapy/utils/spider.py +++ b/scrapy/utils/spider.py @@ -15,7 +15,7 @@ def iterate_spider_output(result): def iter_spider_classes(module): """Return an iterator over all spider classes defined in the given module - that can be instantiated (ie. which have name) + that can be instantiated (i.e. which have name) """ # this needs to be imported here until get rid of the spider manager # singleton in scrapy.spider.spiders diff --git a/sep/sep-003.rst b/sep/sep-003.rst index 184839525..e6357313d 100644 --- a/sep/sep-003.rst +++ b/sep/sep-003.rst @@ -18,7 +18,7 @@ Prerequisites This API proposal relies on the following API: -1. instantiating a item with an item instance as its first argument (ie. +1. instantiating a item with an item instance as its first argument (i.e. ``item2 = MyItem(item1)``) must return a **copy** of the first item instance) 2. items can be instantiated using this syntax: ``item = Item(attr1=value1, @@ -78,7 +78,7 @@ Defining an item containing ItemField's variants2 = ListField(ItemField(Variant), default=[]) It's important to note here that the (perhaps most intuitive) way of defining a -Product-Variant relationship (ie. defining a recursive !ItemField) doesn't +Product-Variant relationship (i.e. defining a recursive !ItemField) doesn't work. For example, this fails to compile: :: diff --git a/sep/sep-013.rst b/sep/sep-013.rst index 5b18b7501..4bc9abd30 100644 --- a/sep/sep-013.rst +++ b/sep/sep-013.rst @@ -59,7 +59,7 @@ Global changes to all middlewares To be discussed: -1. should we support returning deferreds (ie. ``maybeDeferred``) in middleware +1. should we support returning deferreds (i.e. ``maybeDeferred``) in middleware methods? 2. should we pass Twisted Failures instead of exceptions to error methods? diff --git a/sep/sep-021.rst b/sep/sep-021.rst index 628a95dd2..372429791 100644 --- a/sep/sep-021.rst +++ b/sep/sep-021.rst @@ -38,7 +38,7 @@ Goals: * simple to manage: adding or removing extensions should be just a matter of adding or removing lines in a ``scrapy.cfg`` file -* backward compatibility with enabling extension the "old way" (ie. modifying +* backward compatibility with enabling extension the "old way" (i.e. modifying settings directly) Non-goals: From eb21dae5240d2b66feb72940cdd141dba31ecd7a Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Wed, 19 Feb 2020 17:49:42 +0100 Subject: [PATCH 216/313] deprecare sel shortcut in scrapy shell --- scrapy/shell.py | 13 ------------- 1 file changed, 13 deletions(-) diff --git a/scrapy/shell.py b/scrapy/shell.py index a23b04df9..e1b4a024e 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -126,7 +126,6 @@ class Shell(object): self.vars['spider'] = spider self.vars['request'] = request self.vars['response'] = response - self.vars['sel'] = _SelectorProxy(response) if self.inthread: self.vars['fetch'] = self.fetch self.vars['view'] = open_in_browser @@ -192,15 +191,3 @@ def _request_deferred(request): request.callback, request.errback = d.callback, d.errback return d - - -class _SelectorProxy(object): - - def __init__(self, response): - self._proxiedresponse = response - - def __getattr__(self, name): - warnings.warn('"sel" shortcut is deprecated. Use "response.xpath()", ' - '"response.css()" or "response.selector" instead', - category=ScrapyDeprecationWarning, stacklevel=2) - return getattr(self._proxiedresponse.selector, name) From 6972a197073af11bcb582cc03f6286fceda5ca6c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Wed, 19 Feb 2020 18:59:09 +0100 Subject: [PATCH 217/313] Remove unused imports --- scrapy/shell.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scrapy/shell.py b/scrapy/shell.py index e1b4a024e..e22c48dc5 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -5,14 +5,13 @@ See documentation in docs/topics/shell.rst """ import os import signal -import warnings from twisted.internet import threads, defer from twisted.python import threadable from w3lib.url import any_to_uri from scrapy.crawler import Crawler -from scrapy.exceptions import IgnoreRequest, ScrapyDeprecationWarning +from scrapy.exceptions import IgnoreRequest from scrapy.http import Request, Response from scrapy.item import BaseItem from scrapy.settings import Settings From 0f78a591f8796686dc65854c250d6ef5324024aa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Wed, 19 Feb 2020 19:09:39 +0100 Subject: [PATCH 218/313] =?UTF-8?q?Fix=20Flake8-reported=20=E2=80=9CToo=20?= =?UTF-8?q?many=20blank=20lines=E2=80=9D?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scrapy/core/scraper.py | 1 - 1 file changed, 1 deletion(-) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 7b62068f5..41f015017 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -18,7 +18,6 @@ from scrapy.item import BaseItem from scrapy.core.spidermw import SpiderMiddlewareManager - logger = logging.getLogger(__name__) From 91bbc70bc10cf326940eaf53294149444f43fb9e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Marc=20Hern=C3=A1ndez?= <noviluni@gmail.com> Date: Fri, 21 Feb 2020 06:05:31 +0100 Subject: [PATCH 219/313] fix E30X flake8 (#4355) --- pytest.ini | 67 ++++++++++---------- scrapy/core/downloader/tls.py | 1 + scrapy/core/engine.py | 1 + scrapy/responsetypes.py | 1 + scrapy/utils/console.py | 1 + scrapy/utils/gz.py | 1 + tests/test_command_parse.py | 1 - tests/test_crawler.py | 1 + tests/test_dependencies.py | 1 + tests/test_downloadermiddleware_cookies.py | 1 - tests/test_downloadermiddleware_httpcache.py | 1 + tests/test_downloadermiddleware_redirect.py | 4 +- tests/test_exporters.py | 1 + tests/test_http_response.py | 1 + tests/test_item.py | 1 + tests/test_mail.py | 1 + tests/test_pipeline_files.py | 1 - tests/test_pipeline_images.py | 1 - tests/test_pipeline_media.py | 1 + tests/test_responsetypes.py | 1 + tests/test_utils_conf.py | 1 - tests/test_utils_defer.py | 2 + tests/test_utils_deprecate.py | 3 + tests/test_utils_iterators.py | 1 - tests/test_utils_python.py | 3 +- tests/test_utils_request.py | 1 + tests/test_utils_template.py | 1 + tests/test_utils_url.py | 1 + tests/test_webclient.py | 1 + 29 files changed, 58 insertions(+), 45 deletions(-) diff --git a/pytest.ini b/pytest.ini index 552829d4e..0758d2f8b 100644 --- a/pytest.ini +++ b/pytest.ini @@ -47,17 +47,17 @@ flake8-ignore = scrapy/contracts/__init__.py E501 W504 scrapy/contracts/default.py E128 # scrapy/core - scrapy/core/engine.py E501 E128 E127 E306 E502 + scrapy/core/engine.py E501 E128 E127 E502 scrapy/core/scheduler.py E501 - scrapy/core/scraper.py E501 E306 E128 W504 + scrapy/core/scraper.py E501 E128 W504 scrapy/core/spidermw.py E501 E731 E126 E226 scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 - scrapy/core/downloader/tls.py E501 E305 E241 + scrapy/core/downloader/tls.py E501 E241 scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226 scrapy/core/downloader/handlers/__init__.py E501 - scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127 + scrapy/core/downloader/handlers/ftp.py E501 E128 E127 scrapy/core/downloader/handlers/http10.py E501 scrapy/core/downloader/handlers/http11.py E501 scrapy/core/downloader/handlers/s3.py E501 E128 E126 @@ -76,7 +76,7 @@ flake8-ignore = scrapy/extensions/closespider.py E501 E128 E123 scrapy/extensions/corestats.py E501 scrapy/extensions/feedexport.py E128 E501 - scrapy/extensions/httpcache.py E128 E501 E303 + scrapy/extensions/httpcache.py E128 E501 scrapy/extensions/memdebug.py E501 scrapy/extensions/spiderstate.py E501 scrapy/extensions/telnet.py E501 W504 @@ -121,12 +121,11 @@ flake8-ignore = scrapy/utils/asyncio.py E501 scrapy/utils/benchserver.py E501 scrapy/utils/conf.py E402 E501 - scrapy/utils/console.py E306 E305 scrapy/utils/datatypes.py E501 E226 scrapy/utils/decorators.py E501 scrapy/utils/defer.py E501 E128 scrapy/utils/deprecate.py E128 E501 E127 E502 - scrapy/utils/gz.py E305 E501 W504 + scrapy/utils/gz.py E501 W504 scrapy/utils/http.py F403 E226 scrapy/utils/httpobj.py E501 scrapy/utils/iterators.py E501 E701 @@ -161,7 +160,7 @@ flake8-ignore = scrapy/middleware.py E128 E501 scrapy/pqueues.py E501 scrapy/resolver.py E501 - scrapy/responsetypes.py E128 E501 E305 + scrapy/responsetypes.py E128 E501 scrapy/robotstxt.py E501 scrapy/shell.py E501 scrapy/signalmanager.py E501 @@ -175,50 +174,50 @@ flake8-ignore = tests/spiders.py E501 E127 tests/test_closespider.py E501 E127 tests/test_command_fetch.py E501 - tests/test_command_parse.py E501 E128 E303 E226 + tests/test_command_parse.py E501 E128 E226 tests/test_command_shell.py E501 E128 tests/test_commands.py E128 E501 tests/test_contracts.py E501 E128 tests/test_crawl.py E501 E741 E265 - tests/test_crawler.py F841 E306 E501 - tests/test_dependencies.py F841 E501 E305 + tests/test_crawler.py F841 E501 + tests/test_dependencies.py F841 E501 tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 - tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126 + tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126 tests/test_downloadermiddleware_decompression.py E127 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 - tests/test_downloadermiddleware_httpcache.py E501 E305 + tests/test_downloadermiddleware_httpcache.py E501 tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123 tests/test_downloadermiddleware_httpproxy.py E501 E128 - tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305 - tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126 + tests/test_downloadermiddleware_redirect.py E501 E128 E127 + tests/test_downloadermiddleware_retry.py E501 E128 E251 E126 tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 tests/test_dupefilters.py E221 E501 E741 E128 E124 tests/test_engine.py E401 E501 E128 - tests/test_exporters.py E501 E731 E306 E128 E124 + tests/test_exporters.py E501 E731 E128 E124 tests/test_extension_telnet.py F841 tests/test_feedexport.py E501 F841 E241 tests/test_http_cookies.py E501 tests/test_http_headers.py E501 tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 - tests/test_http_response.py E501 E301 E128 E265 - tests/test_item.py E701 E128 F841 E306 + tests/test_http_response.py E501 E128 E265 + tests/test_item.py E701 E128 F841 tests/test_link.py E501 tests/test_linkextractors.py E501 E128 E124 - tests/test_loader.py E501 E731 E303 E741 E128 E117 E241 + tests/test_loader.py E501 E731 E741 E128 E117 E241 tests/test_logformatter.py E128 E501 E122 - tests/test_mail.py E128 E501 E305 + tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 tests/test_pipeline_crawl.py E131 E501 E128 E126 - tests/test_pipeline_files.py E501 E303 E272 E226 - tests/test_pipeline_images.py F841 E501 E303 - tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502 + tests/test_pipeline_files.py E501 E272 E226 + tests/test_pipeline_images.py F841 E501 + tests/test_pipeline_media.py E501 E741 E731 E128 E502 tests/test_proxy_connect.py E501 E741 tests/test_request_cb_kwargs.py E501 - tests/test_responsetypes.py E501 E305 + tests/test_responsetypes.py E501 tests/test_robotstxt_interface.py E501 E501 tests/test_scheduler.py E501 E126 E123 tests/test_selector.py E501 E127 @@ -230,24 +229,22 @@ flake8-ignore = tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 tests/test_squeues.py E501 E701 E741 tests/test_utils_asyncio.py E501 - tests/test_utils_conf.py E501 E303 E128 + tests/test_utils_conf.py E501 E128 tests/test_utils_curl.py E501 - tests/test_utils_datatypes.py E402 E501 E305 - tests/test_utils_defer.py E306 E501 F841 E226 - tests/test_utils_deprecate.py F841 E306 E501 + tests/test_utils_datatypes.py E402 E501 + tests/test_utils_defer.py E501 F841 E226 + tests/test_utils_deprecate.py F841 E501 tests/test_utils_http.py E501 E128 W504 - tests/test_utils_iterators.py E501 E128 E129 E303 E241 + tests/test_utils_iterators.py E501 E128 E129 E241 tests/test_utils_log.py E741 E226 - tests/test_utils_python.py E501 E303 E731 E701 E305 + tests/test_utils_python.py E501 E731 E701 tests/test_utils_reqser.py E501 E128 - tests/test_utils_request.py E501 E128 E305 + tests/test_utils_request.py E501 E128 tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 E731 E226 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_spider.py E305 - tests/test_utils_template.py E305 - tests/test_utils_url.py E501 E127 E305 E211 E125 E501 E226 E241 E126 E123 - tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126 + tests/test_utils_url.py E501 E127 E211 E125 E501 E226 E241 E126 E123 + tests/test_webclient.py E501 E128 E122 E402 E226 E241 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 tests/test_spiderloader/__init__.py E128 E501 diff --git a/scrapy/core/downloader/tls.py b/scrapy/core/downloader/tls.py index 4ed482058..a1c881d5e 100644 --- a/scrapy/core/downloader/tls.py +++ b/scrapy/core/downloader/tls.py @@ -89,4 +89,5 @@ class ScrapyClientTLSOptions(ClientTLSOptions): 'from host "{}" (exception: {})'.format( self._hostnameASCII, repr(e))) + DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT') diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 829e69993..6ab8cde6b 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -230,6 +230,7 @@ class ExecutionEngine(object): def _download(self, request, spider): slot = self.slot slot.add_request(request) + def _on_success(response): assert isinstance(response, (Response, Request)) if isinstance(response, Response): diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 91d309147..64bf93e86 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -116,4 +116,5 @@ class ResponseTypes(object): cls = self.from_body(body) return cls + responsetypes = ResponseTypes() diff --git a/scrapy/utils/console.py b/scrapy/utils/console.py index 7eb40f0ce..c7a2ace88 100644 --- a/scrapy/utils/console.py +++ b/scrapy/utils/console.py @@ -54,6 +54,7 @@ def _embed_standard_shell(namespace={}, banner=''): else: import rlcompleter # noqa: F401 readline.parse_and_bind("tab:complete") + @wraps(_embed_standard_shell) def wrapper(namespace=namespace, banner=''): code.interact(banner=banner, local=namespace) diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index 9672e28da..c291ae237 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -42,6 +42,7 @@ def gunzip(data): raise return b''.join(output_list) + _is_gzipped = re.compile(br'^application/(x-)?gzip\b', re.I).search _is_octetstream = re.compile(br'^(application|binary)/octet-stream\b', re.I).search diff --git a/tests/test_command_parse.py b/tests/test_command_parse.py index b7035fdff..8a54d2c74 100644 --- a/tests/test_command_parse.py +++ b/tests/test_command_parse.py @@ -147,7 +147,6 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} self.url('/html')]) self.assertIn("DEBUG: It Works!", _textmode(stderr)) - @defer.inlineCallbacks def test_pipelines(self): _, _, stderr = yield self.execute(['--spider', self.spider_name, diff --git a/tests/test_crawler.py b/tests/test_crawler.py index f8fa26def..7bd76601d 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -107,6 +107,7 @@ class CrawlerLoggingTestCase(unittest.TestCase): def test_spider_custom_settings_log_level(self): log_file = self.mktemp() + class MySpider(scrapy.Spider): name = 'spider' custom_settings = { diff --git a/tests/test_dependencies.py b/tests/test_dependencies.py index e31ccd9b5..a169acbe6 100644 --- a/tests/test_dependencies.py +++ b/tests/test_dependencies.py @@ -13,5 +13,6 @@ class ScrapyUtilsTest(unittest.TestCase): installed_version = [int(x) for x in module.__version__.split('.')[:2]] assert installed_version >= [0, 6], "OpenSSL >= 0.6 required" + if __name__ == "__main__": unittest.main() diff --git a/tests/test_downloadermiddleware_cookies.py b/tests/test_downloadermiddleware_cookies.py index 04884fb78..051f66680 100644 --- a/tests/test_downloadermiddleware_cookies.py +++ b/tests/test_downloadermiddleware_cookies.py @@ -145,7 +145,6 @@ class CookiesMiddlewareTest(TestCase): {'name': 'C3', 'value': 'value3', 'path': '/foo', 'domain': 'scrapytest.org'}, {'name': 'C4', 'value': 'value4', 'path': '/foo', 'domain': 'scrapy.org'}] - req = Request('http://scrapytest.org/', cookies=cookies) self.mw.process_request(req, self.spider) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 9401dd66d..9b77c97a8 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -501,5 +501,6 @@ class RFC2616PolicyTest(DefaultStorageTest): self.assertEqualResponse(res1, res2) assert 'cached' in res2.flags + if __name__ == '__main__': unittest.main() diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index e0f145d0e..053e26fc3 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -68,7 +68,6 @@ class RedirectMiddlewareTest(unittest.TestCase): assert isinstance(r, Response) assert r is rsp - def test_redirect_302(self): url = 'http://www.example.com/302' url2 = 'http://www.example.com/redirected2' @@ -122,7 +121,6 @@ class RedirectMiddlewareTest(unittest.TestCase): del rsp.headers['Location'] assert self.mw.process_response(req, rsp, self.spider) is rsp - def test_max_redirect_times(self): self.mw.max_redirect_times = 1 req = Request('http://scrapytest.org/302') @@ -178,6 +176,7 @@ class RedirectMiddlewareTest(unittest.TestCase): def test_request_meta_handling(self): url = 'http://www.example.com/301' url2 = 'http://www.example.com/redirected' + def _test_passthrough(req): rsp = Response(url, headers={'Location': url2}, status=301, request=req) r = self.mw.process_response(req, rsp, self.spider) @@ -316,5 +315,6 @@ class MetaRefreshMiddlewareTest(unittest.TestCase): response = mw.process_response(req, rsp, self.spider) assert isinstance(response, Response) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_exporters.py b/tests/test_exporters.py index 5d1f5c182..6e2507508 100644 --- a/tests/test_exporters.py +++ b/tests/test_exporters.py @@ -312,6 +312,7 @@ class XmlItemExporterTest(BaseItemExporterTest): for child in children] else: return [(elem.tag, [(elem.text, ())])] + def xmlsplit(xmlcontent): doc = lxml.etree.fromstring(xmlcontent) return xmltuple(doc) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 0dc603923..be17dfd6b 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -182,6 +182,7 @@ class BaseResponseTest(unittest.TestCase): def test_follow_whitespace_link(self): self._assert_followed_url(Link('http://example.com/foo '), 'http://example.com/foo%20') + def test_follow_flags(self): res = self.response_class('http://example.com/') fol = res.follow('http://example.com/', flags=['cached', 'allowed']) diff --git a/tests/test_item.py b/tests/test_item.py index 30463a0f5..823bf1ced 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -259,6 +259,7 @@ class ItemTest(unittest.TestCase): with catch_warnings(record=True) as warnings: item = Item() self.assertEqual(len(warnings), 0) + class SubclassedItem(Item): pass subclassed_item = SubclassedItem() diff --git a/tests/test_mail.py b/tests/test_mail.py index ddb0f1e70..f5cb81a8b 100644 --- a/tests/test_mail.py +++ b/tests/test_mail.py @@ -121,5 +121,6 @@ class MailSenderTest(unittest.TestCase): self.assertEqual(text.get_charset(), Charset('utf-8')) self.assertEqual(attach.get_payload(decode=True).decode('utf-8'), body) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index e5bad2ed0..88ce1cf18 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -286,7 +286,6 @@ class FilesPipelineTestCaseCustomSettings(unittest.TestCase): self.assertEqual(pipeline.files_result_field, "this") self.assertEqual(pipeline.files_urls_field, "that") - def test_user_defined_subclass_default_key_names(self): """Test situation when user defines subclass of FilesPipeline, but uses attribute names for default pipeline (without prefixing diff --git a/tests/test_pipeline_images.py b/tests/test_pipeline_images.py index 7f1cb4a11..5018d6802 100644 --- a/tests/test_pipeline_images.py +++ b/tests/test_pipeline_images.py @@ -177,7 +177,6 @@ class ImagesPipelineTestCaseCustomSettings(unittest.TestCase): IMAGES_RESULT_FIELD='images' ) - def setUp(self): self.tempdir = mkdtemp() diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index 1fcc5799e..d369e147d 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -304,6 +304,7 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): return response rsp1 = Response('http://url') + def rsp1_func(): dfd = Deferred().addCallback(_check_downloading) reactor.callLater(.1, dfd.callback, rsp1) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index d5a3371ab..8cdf7a176 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -90,5 +90,6 @@ class ResponseTypesTest(unittest.TestCase): # check that mime.types files shipped with scrapy are loaded self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test') + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index 02d8ba51e..61e110845 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -83,7 +83,6 @@ class BuildComponentListTest(unittest.TestCase): self.assertRaises(ValueError, build_component_list, {}, d, convert=lambda x: x) - class UtilsConfTestCase(unittest.TestCase): def test_arglist_to_dict(self): diff --git a/tests/test_utils_defer.py b/tests/test_utils_defer.py index dfbe71ae2..89b5fb4fb 100644 --- a/tests/test_utils_defer.py +++ b/tests/test_utils_defer.py @@ -9,6 +9,7 @@ from scrapy.utils.defer import mustbe_deferred, process_chain, \ class MustbeDeferredTest(unittest.TestCase): def test_success_function(self): steps = [] + def _append(v): steps.append(v) return steps @@ -20,6 +21,7 @@ class MustbeDeferredTest(unittest.TestCase): def test_unfired_deferred(self): steps = [] + def _append(v): steps.append(v) dfd = defer.Deferred() diff --git a/tests/test_utils_deprecate.py b/tests/test_utils_deprecate.py index 159ef8f25..b3a90d314 100644 --- a/tests/test_utils_deprecate.py +++ b/tests/test_utils_deprecate.py @@ -110,6 +110,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): # ignore subclassing warnings with warnings.catch_warnings(): warnings.simplefilter('ignore', ScrapyDeprecationWarning) + class UserClass(Deprecated): pass @@ -233,6 +234,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): with warnings.catch_warnings(record=True) as w: AlsoDeprecated() + class UserClass(AlsoDeprecated): pass @@ -247,6 +249,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): with mock.patch('inspect.stack', side_effect=IndexError): with warnings.catch_warnings(record=True) as w: DeprecatedName = create_deprecated_class('DeprecatedName', NewName) + class SubClass(DeprecatedName): pass diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index 9776dfb2a..33fc4d570 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -387,7 +387,6 @@ class TestHelper(unittest.TestCase): self.assertTrue(type(r1) is type(r2)) self.assertTrue(type(r1) is not type(r3)) - def _assert_type_and_value(self, a, b, obj): self.assertTrue(type(a) is type(b), 'Got {}, expected {} for {!r}'.format(type(a), type(b), obj)) diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index b79e0ac1c..4202e8c89 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -104,7 +104,6 @@ class BinaryIsTextTest(unittest.TestCase): assert not binary_is_text(b"\x02\xa3") - class UtilsPythonTestCase(unittest.TestCase): def test_equal_attributes(self): @@ -215,7 +214,6 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertEqual( get_func_args(operator.itemgetter(2), stripself=True), ['obj']) - def test_without_none_values(self): self.assertEqual(without_none_values([1, None, 3, 4]), [1, 3, 4]) self.assertEqual(without_none_values((1, None, 3, 4)), (1, 3, 4)) @@ -223,5 +221,6 @@ class UtilsPythonTestCase(unittest.TestCase): without_none_values({'one': 1, 'none': None, 'three': 3, 'four': 4}), {'one': 1, 'three': 3, 'four': 4}) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_request.py b/tests/test_utils_request.py index 3e664fc74..45f0f59e4 100644 --- a/tests/test_utils_request.py +++ b/tests/test_utils_request.py @@ -83,5 +83,6 @@ class UtilsRequestTest(unittest.TestCase): request_httprepr(Request("file:///tmp/foo.txt")) request_httprepr(Request("ftp://localhost/tmp/foo.txt")) + if __name__ == "__main__": unittest.main() diff --git a/tests/test_utils_template.py b/tests/test_utils_template.py index 40b733233..5a52dd695 100644 --- a/tests/test_utils_template.py +++ b/tests/test_utils_template.py @@ -38,5 +38,6 @@ class UtilsRenderTemplateFileTestCase(unittest.TestCase): os.remove(render_path) assert not os.path.exists(render_path) # Failure of test iself + if '__main__' == __name__: unittest.main() diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 21e9a056a..9f1acbc75 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -201,6 +201,7 @@ def create_skipped_scheme_t(args): assert url.startswith(args[1]) return do_expected + for k, args in enumerate ([ ('/index', 'file://'), ('/index.html', 'file://'), diff --git a/tests/test_webclient.py b/tests/test_webclient.py index 746367b41..b602a3ea0 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -294,6 +294,7 @@ class WebClientTestCase(unittest.TestCase): finished = self.assertFailure( getPage(self.getURL("wait"), timeout=0.000001), defer.TimeoutError) + def cleanup(passthrough): # Clean up the server which is hanging around not doing # anything. From 6fb85951ce3843156a801e71441a4a3e387588e2 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Thu, 20 Feb 2020 16:32:58 +0100 Subject: [PATCH 220/313] fix E22X flake8 --- pytest.ini | 48 ++++++++++----------- scrapy/commands/parse.py | 6 +-- scrapy/core/downloader/webclient.py | 2 +- scrapy/core/spidermw.py | 6 +-- scrapy/downloadermiddlewares/ajaxcrawl.py | 2 +- scrapy/exporters.py | 4 +- scrapy/linkextractors/lxmlhtml.py | 2 +- scrapy/settings/default_settings.py | 4 +- scrapy/spiderloader.py | 2 +- scrapy/utils/datatypes.py | 6 +-- scrapy/utils/http.py | 2 +- scrapy/utils/misc.py | 2 +- scrapy/utils/reactor.py | 2 +- tests/pipelines.py | 4 +- tests/test_command_parse.py | 8 ++-- tests/test_downloader_handlers.py | 4 +- tests/test_dupefilters.py | 10 ++--- tests/test_pipeline_files.py | 2 +- tests/test_spidermiddleware.py | 2 +- tests/test_spidermiddleware_output_chain.py | 2 +- tests/test_utils_defer.py | 2 +- tests/test_utils_log.py | 2 +- tests/test_utils_signal.py | 2 +- tests/test_utils_url.py | 2 +- tests/test_webclient.py | 10 ++--- 25 files changed, 69 insertions(+), 69 deletions(-) diff --git a/pytest.ini b/pytest.ini index 0758d2f8b..7806620d5 100644 --- a/pytest.ini +++ b/pytest.ini @@ -37,7 +37,7 @@ flake8-ignore = scrapy/commands/edit.py E501 scrapy/commands/fetch.py E401 E501 E128 E731 scrapy/commands/genspider.py E128 E501 E502 - scrapy/commands/parse.py E128 E501 E731 E226 + scrapy/commands/parse.py E128 E501 E731 scrapy/commands/runspider.py E501 scrapy/commands/settings.py E128 scrapy/commands/shell.py E128 E501 E502 @@ -50,19 +50,19 @@ flake8-ignore = scrapy/core/engine.py E501 E128 E127 E502 scrapy/core/scheduler.py E501 scrapy/core/scraper.py E501 E128 W504 - scrapy/core/spidermw.py E501 E731 E126 E226 + scrapy/core/spidermw.py E501 E731 E126 scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 scrapy/core/downloader/tls.py E501 E241 - scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226 + scrapy/core/downloader/webclient.py E731 E501 E128 E126 scrapy/core/downloader/handlers/__init__.py E501 scrapy/core/downloader/handlers/ftp.py E501 E128 E127 scrapy/core/downloader/handlers/http10.py E501 scrapy/core/downloader/handlers/http11.py E501 scrapy/core/downloader/handlers/s3.py E501 E128 E126 # scrapy/downloadermiddlewares - scrapy/downloadermiddlewares/ajaxcrawl.py E501 E226 + scrapy/downloadermiddlewares/ajaxcrawl.py E501 scrapy/downloadermiddlewares/decompression.py E501 scrapy/downloadermiddlewares/defaultheaders.py E501 scrapy/downloadermiddlewares/httpcache.py E501 E126 @@ -91,7 +91,7 @@ flake8-ignore = scrapy/http/response/text.py E501 E128 E124 # scrapy/linkextractors scrapy/linkextractors/__init__.py E731 E501 E402 W504 - scrapy/linkextractors/lxmlhtml.py E501 E731 E226 + scrapy/linkextractors/lxmlhtml.py E501 E731 # scrapy/loader scrapy/loader/__init__.py E501 E128 scrapy/loader/processors.py E501 @@ -105,7 +105,7 @@ flake8-ignore = scrapy/selector/unified.py E501 E111 # scrapy/settings scrapy/settings/__init__.py E501 - scrapy/settings/default_settings.py E501 E114 E116 E226 + scrapy/settings/default_settings.py E501 E114 E116 scrapy/settings/deprecated.py E501 # scrapy/spidermiddlewares scrapy/spidermiddlewares/httperror.py E501 @@ -121,21 +121,21 @@ flake8-ignore = scrapy/utils/asyncio.py E501 scrapy/utils/benchserver.py E501 scrapy/utils/conf.py E402 E501 - scrapy/utils/datatypes.py E501 E226 + scrapy/utils/datatypes.py E501 scrapy/utils/decorators.py E501 scrapy/utils/defer.py E501 E128 scrapy/utils/deprecate.py E128 E501 E127 E502 scrapy/utils/gz.py E501 W504 - scrapy/utils/http.py F403 E226 + scrapy/utils/http.py F403 scrapy/utils/httpobj.py E501 scrapy/utils/iterators.py E501 E701 scrapy/utils/log.py E128 E501 scrapy/utils/markup.py F403 - scrapy/utils/misc.py E501 E226 + scrapy/utils/misc.py E501 scrapy/utils/multipart.py F403 scrapy/utils/project.py E501 scrapy/utils/python.py E501 - scrapy/utils/reactor.py E226 E501 + scrapy/utils/reactor.py E501 scrapy/utils/reqser.py E501 scrapy/utils/request.py E127 E501 scrapy/utils/response.py E501 E128 @@ -151,7 +151,7 @@ flake8-ignore = scrapy/crawler.py E501 scrapy/dupefilters.py E501 E202 scrapy/exceptions.py E501 - scrapy/exporters.py E501 E226 + scrapy/exporters.py E501 scrapy/interfaces.py E501 scrapy/item.py E501 E128 scrapy/link.py E501 @@ -164,24 +164,24 @@ flake8-ignore = scrapy/robotstxt.py E501 scrapy/shell.py E501 scrapy/signalmanager.py E501 - scrapy/spiderloader.py E225 F841 E501 E126 + scrapy/spiderloader.py F841 E501 E126 scrapy/squeues.py E128 scrapy/statscollectors.py E501 # tests tests/__init__.py E402 E501 tests/mockserver.py E401 E501 E126 E123 - tests/pipelines.py F841 E226 + tests/pipelines.py F841 tests/spiders.py E501 E127 tests/test_closespider.py E501 E127 tests/test_command_fetch.py E501 - tests/test_command_parse.py E501 E128 E226 + tests/test_command_parse.py E501 E128 tests/test_command_shell.py E501 E128 tests/test_commands.py E128 E501 tests/test_contracts.py E501 E128 tests/test_crawl.py E501 E741 E265 tests/test_crawler.py F841 E501 tests/test_dependencies.py F841 E501 - tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123 + tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E701 E126 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126 @@ -195,7 +195,7 @@ flake8-ignore = tests/test_downloadermiddleware_retry.py E501 E128 E251 E126 tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 - tests/test_dupefilters.py E221 E501 E741 E128 E124 + tests/test_dupefilters.py E501 E741 E128 E124 tests/test_engine.py E401 E501 E128 tests/test_exporters.py E501 E731 E128 E124 tests/test_extension_telnet.py F841 @@ -212,7 +212,7 @@ flake8-ignore = tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 tests/test_pipeline_crawl.py E131 E501 E128 E126 - tests/test_pipeline_files.py E501 E272 E226 + tests/test_pipeline_files.py E501 E272 tests/test_pipeline_images.py F841 E501 tests/test_pipeline_media.py E501 E741 E731 E128 E502 tests/test_proxy_connect.py E501 E741 @@ -222,29 +222,29 @@ flake8-ignore = tests/test_scheduler.py E501 E126 E123 tests/test_selector.py E501 E127 tests/test_spider.py E501 - tests/test_spidermiddleware.py E501 E226 + tests/test_spidermiddleware.py E501 tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 tests/test_spidermiddleware_offsite.py E501 E128 E111 - tests/test_spidermiddleware_output_chain.py E501 E226 + tests/test_spidermiddleware_output_chain.py E501 tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 tests/test_squeues.py E501 E701 E741 tests/test_utils_asyncio.py E501 tests/test_utils_conf.py E501 E128 tests/test_utils_curl.py E501 tests/test_utils_datatypes.py E402 E501 - tests/test_utils_defer.py E501 F841 E226 + tests/test_utils_defer.py E501 F841 tests/test_utils_deprecate.py F841 E501 tests/test_utils_http.py E501 E128 W504 tests/test_utils_iterators.py E501 E128 E129 E241 - tests/test_utils_log.py E741 E226 + tests/test_utils_log.py E741 tests/test_utils_python.py E501 E731 E701 tests/test_utils_reqser.py E501 E128 tests/test_utils_request.py E501 E128 tests/test_utils_response.py E501 - tests/test_utils_signal.py E741 F841 E731 E226 + tests/test_utils_signal.py E741 F841 E731 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_url.py E501 E127 E211 E125 E501 E226 E241 E126 E123 - tests/test_webclient.py E501 E128 E122 E402 E226 E241 E123 E126 + tests/test_utils_url.py E501 E127 E211 E125 E501 E241 E126 E123 + tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 tests/test_spiderloader/__init__.py E128 E501 diff --git a/scrapy/commands/parse.py b/scrapy/commands/parse.py index ff6f1d8cd..3ef8ddcb3 100644 --- a/scrapy/commands/parse.py +++ b/scrapy/commands/parse.py @@ -80,7 +80,7 @@ class Command(ScrapyCommand): else: items = self.items.get(lvl, []) - print("# Scraped Items ", "-"*60) + print("# Scraped Items ", "-" * 60) display.pprint([dict(x) for x in items], colorize=colour) def print_requests(self, lvl=None, colour=True): @@ -92,14 +92,14 @@ class Command(ScrapyCommand): else: requests = self.requests.get(lvl, []) - print("# Requests ", "-"*65) + print("# Requests ", "-" * 65) display.pprint(requests, colorize=colour) def print_results(self, opts): colour = not opts.nocolour if opts.verbose: - for level in range(1, self.max_level+1): + for level in range(1, self.max_level + 1): print('\n>>> DEPTH LEVEL: %s <<<' % level) if not opts.noitems: self.print_items(level, colour) diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index fc796e8bb..a71dc5fb3 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -140,7 +140,7 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): self.headers['Content-Length'] = 0 def _build_response(self, body, request): - request.meta['download_latency'] = self.headers_time-self.start_time + request.meta['download_latency'] = self.headers_time - self.start_time status = int(self.status) headers = Headers(self.response_headers) respcls = responsetypes.from_args(headers=headers, url=self._url) diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index dd9b3c376..87d08cab7 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -82,7 +82,7 @@ class SpiderMiddlewareManager(MiddlewareManager): if _isiterable(result): # stop exception handling by handing control over to the # process_spider_output chain if an iterable has been returned - return process_spider_output(result, method_index+1) + return process_spider_output(result, method_index + 1) elif result is None: continue else: @@ -103,12 +103,12 @@ class SpiderMiddlewareManager(MiddlewareManager): # might fail directly if the output value is not a generator result = method(response=response, result=result, spider=spider) except Exception as ex: - exception_result = process_spider_exception(Failure(ex), method_index+1) + exception_result = process_spider_exception(Failure(ex), method_index + 1) if isinstance(exception_result, Failure): raise return exception_result if _isiterable(result): - result = _evaluate_iterable(result, method_index+1, recovered) + result = _evaluate_iterable(result, method_index + 1, recovered) else: msg = "Middleware {} must return an iterable, got {}" raise _InvalidOutput(msg.format(_fname(method), type(result))) diff --git a/scrapy/downloadermiddlewares/ajaxcrawl.py b/scrapy/downloadermiddlewares/ajaxcrawl.py index 7a140fcad..16b046e99 100644 --- a/scrapy/downloadermiddlewares/ajaxcrawl.py +++ b/scrapy/downloadermiddlewares/ajaxcrawl.py @@ -47,7 +47,7 @@ class AjaxCrawlMiddleware(object): return response # scrapy already handles #! links properly - ajax_crawl_request = request.replace(url=request.url+'#!') + ajax_crawl_request = request.replace(url=request.url + '#!') logger.debug("Downloading AJAX crawlable %(ajax_crawl_request)s instead of %(request)s", {'ajax_crawl_request': ajax_crawl_request, 'request': request}, extra={'spider': spider}) diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 1a3c9345f..96416f075 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -173,12 +173,12 @@ class XmlItemExporter(BaseItemExporter): if hasattr(serialized_value, 'items'): self._beautify_newline() for subname, value in serialized_value.items(): - self._export_xml_field(subname, value, depth=depth+1) + self._export_xml_field(subname, value, depth=depth + 1) self._beautify_indent(depth=depth) elif is_listlike(serialized_value): self._beautify_newline() for value in serialized_value: - self._export_xml_field('value', value, depth=depth+1) + self._export_xml_field('value', value, depth=depth + 1) self._beautify_indent(depth=depth) elif isinstance(serialized_value, str): self.xg.characters(serialized_value) diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index f5ef56ea4..ab82e1915 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -22,7 +22,7 @@ _collect_string_content = etree.XPath("string()") def _nons(tag): if isinstance(tag, str): - if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE)+1] == XHTML_NAMESPACE: + if tag[0] == '{' and tag[1:len(XHTML_NAMESPACE) + 1] == XHTML_NAMESPACE: return tag.split('}')[-1] return tag diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index fc7b62e78..f8a0457ce 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -75,8 +75,8 @@ DOWNLOAD_HANDLERS_BASE = { DOWNLOAD_TIMEOUT = 180 # 3mins -DOWNLOAD_MAXSIZE = 1024*1024*1024 # 1024m -DOWNLOAD_WARNSIZE = 32*1024*1024 # 32m +DOWNLOAD_MAXSIZE = 1024 * 1024 * 1024 # 1024m +DOWNLOAD_WARNSIZE = 32 * 1024 * 1024 # 32m DOWNLOAD_FAIL_ON_DATALOSS = True diff --git a/scrapy/spiderloader.py b/scrapy/spiderloader.py index 3beca4060..048e84e4f 100644 --- a/scrapy/spiderloader.py +++ b/scrapy/spiderloader.py @@ -28,7 +28,7 @@ class SpiderLoader(object): module=mod, cls=cls, name=name) for (mod, cls) in locations) for name, locations in self._found.items() - if len(locations)>1] + if len(locations) > 1] if dupes: msg = ("There are several spiders with the same name:\n\n" "{}\n\n This can cause unexpected behavior.".format( diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index a52bbc70e..b07f995cf 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -175,12 +175,12 @@ class SiteNode(object): node.parent = self def to_string(self, level=0): - s = "%s%s\n" % (' '*level, self.url) + s = "%s%s\n" % (' ' * level, self.url) if self.itemnames: for n in self.itemnames: - s += "%sScraped: %s\n" % (' '*(level+1), n) + s += "%sScraped: %s\n" % (' ' * (level + 1), n) for node in self.children: - s += node.to_string(level+1) + s += node.to_string(level + 1) return s diff --git a/scrapy/utils/http.py b/scrapy/utils/http.py index bab262393..ceb3f0509 100644 --- a/scrapy/utils/http.py +++ b/scrapy/utils/http.py @@ -32,5 +32,5 @@ def decode_chunked_transfer(chunked_body): break size = int(h, 16) body += t[:size] - t = t[size+2:] + t = t[size + 2:] return body diff --git a/scrapy/utils/misc.py b/scrapy/utils/misc.py index a3e55d6ea..52cfba208 100644 --- a/scrapy/utils/misc.py +++ b/scrapy/utils/misc.py @@ -46,7 +46,7 @@ def load_object(path): except ValueError: raise ValueError("Error loading object '%s': not a full path" % path) - module, name = path[:dot], path[dot+1:] + module, name = path[:dot], path[dot + 1:] mod = import_module(module) try: diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 80f52a4ef..6513e06c9 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -16,7 +16,7 @@ def listen_tcp(portrange, host, factory): return reactor.listenTCP(portrange, factory, interface=host) if len(portrange) == 1: return reactor.listenTCP(portrange[0], factory, interface=host) - for x in range(portrange[0], portrange[1]+1): + for x in range(portrange[0], portrange[1] + 1): try: return reactor.listenTCP(x, factory, interface=host) except error.CannotListenError: diff --git a/tests/pipelines.py b/tests/pipelines.py index d7d3b5259..de4894c32 100644 --- a/tests/pipelines.py +++ b/tests/pipelines.py @@ -6,7 +6,7 @@ Some pipelines used for testing class ZeroDivisionErrorPipeline(object): def open_spider(self, spider): - a = 1/0 + a = 1 / 0 def process_item(self, item, spider): return item @@ -15,4 +15,4 @@ class ZeroDivisionErrorPipeline(object): class ProcessWithZeroDivisionErrorPipiline(object): def process_item(self, item, spider): - 1/0 + 1 / 0 diff --git a/tests/test_command_parse.py b/tests/test_command_parse.py index 8a54d2c74..5bf92b71a 100644 --- a/tests/test_command_parse.py +++ b/tests/test_command_parse.py @@ -182,7 +182,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_crawlspider_matching_rule_callback_set(self): """If a rule matches the URL, use it's defined callback.""" status, out, stderr = yield self.execute( - ['--spider', 'goodcrawl'+self.spider_name, '-r', self.url('/html')] + ['--spider', 'goodcrawl' + self.spider_name, '-r', self.url('/html')] ) self.assertIn("""[{}, {'foo': 'bar'}]""", _textmode(out)) @@ -190,7 +190,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_crawlspider_matching_rule_default_callback(self): """If a rule match but it has no callback set, use the 'parse' callback.""" status, out, stderr = yield self.execute( - ['--spider', 'goodcrawl'+self.spider_name, '-r', self.url('/text')] + ['--spider', 'goodcrawl' + self.spider_name, '-r', self.url('/text')] ) self.assertIn("""[{}, {'nomatch': 'default'}]""", _textmode(out)) @@ -206,7 +206,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} @defer.inlineCallbacks def test_crawlspider_missing_callback(self): status, out, stderr = yield self.execute( - ['--spider', 'badcrawl'+self.spider_name, '-r', self.url('/html')] + ['--spider', 'badcrawl' + self.spider_name, '-r', self.url('/html')] ) self.assertRegex(_textmode(out), r"""# Scraped Items -+\n\[\]""") @@ -214,7 +214,7 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1} def test_crawlspider_no_matching_rule(self): """The requested URL has no matching rule, so no items should be scraped""" status, out, stderr = yield self.execute( - ['--spider', 'badcrawl'+self.spider_name, '-r', self.url('/enc-gb18030')] + ['--spider', 'badcrawl' + self.spider_name, '-r', self.url('/enc-gb18030')] ) self.assertRegex(_textmode(out), r"""# Scraped Items -+\n\[\]""") self.assertIn("""Cannot find a rule that matches""", _textmode(stderr)) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 8d95d7cac..29d06bab4 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -348,7 +348,7 @@ class HttpTestCase(unittest.TestCase): return self.download_request(request, Spider('foo')).addCallback(_test) def test_payload(self): - body = b'1'*100 # PayloadResource requires body length to be 100 + body = b'1' * 100 # PayloadResource requires body length to be 100 request = Request(self.getURL('payload'), method='POST', body=body) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) @@ -812,7 +812,7 @@ class S3TestCase(unittest.TestCase): def test_request_signing1(self): # gets an object from the johnsmith bucket. - date ='Tue, 27 Mar 2007 19:36:42 +0000' + date = 'Tue, 27 Mar 2007 19:36:42 +0000' req = Request('s3://johnsmith/photos/puppy.jpg', headers={'Date': date}) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) diff --git a/tests/test_dupefilters.py b/tests/test_dupefilters.py index 88ce9627f..9e24d86dd 100644 --- a/tests/test_dupefilters.py +++ b/tests/test_dupefilters.py @@ -43,7 +43,7 @@ class RFPDupeFilterTest(unittest.TestCase): def test_df_from_crawler_scheduler(self): settings = {'DUPEFILTER_DEBUG': True, - 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} crawler = get_crawler(settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) self.assertTrue(scheduler.df.debug) @@ -51,14 +51,14 @@ class RFPDupeFilterTest(unittest.TestCase): def test_df_from_settings_scheduler(self): settings = {'DUPEFILTER_DEBUG': True, - 'DUPEFILTER_CLASS': __name__ + '.FromSettingsRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromSettingsRFPDupeFilter'} crawler = get_crawler(settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) self.assertTrue(scheduler.df.debug) self.assertEqual(scheduler.df.method, 'from_settings') def test_df_direct_scheduler(self): - settings = {'DUPEFILTER_CLASS': __name__ + '.DirectDupeFilter'} + settings = {'DUPEFILTER_CLASS': __name__ + '.DirectDupeFilter'} crawler = get_crawler(settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) self.assertEqual(scheduler.df.method, 'n/a') @@ -162,7 +162,7 @@ class RFPDupeFilterTest(unittest.TestCase): def test_log(self): with LogCapture() as l: settings = {'DUPEFILTER_DEBUG': False, - 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} crawler = get_crawler(SimpleSpider, settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) spider = SimpleSpider.from_crawler(crawler) @@ -187,7 +187,7 @@ class RFPDupeFilterTest(unittest.TestCase): def test_log_debug(self): with LogCapture() as l: settings = {'DUPEFILTER_DEBUG': True, - 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} + 'DUPEFILTER_CLASS': __name__ + '.FromCrawlerRFPDupeFilter'} crawler = get_crawler(SimpleSpider, settings_dict=settings) scheduler = Scheduler.from_crawler(crawler) spider = SimpleSpider.from_crawler(crawler) diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index 88ce1cf18..799782647 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -359,7 +359,7 @@ class TestGCSFilesStore(unittest.TestCase): self.assertIn('checksum', s) self.assertEqual(s['checksum'], 'zc2oVgXkbQr2EQdSdw3OPA==') u = urlparse(uri) - content, acl, blob = get_gcs_content_and_delete(u.hostname, u.path[1:]+path) + content, acl, blob = get_gcs_content_and_delete(u.hostname, u.path[1:] + path) self.assertEqual(content, data) self.assertEqual(blob.metadata, {'foo': 'bar'}) self.assertEqual(blob.cache_control, GCSFilesStore.CACHE_CONTROL) diff --git a/tests/test_spidermiddleware.py b/tests/test_spidermiddleware.py index 55d665e79..78e926adc 100644 --- a/tests/test_spidermiddleware.py +++ b/tests/test_spidermiddleware.py @@ -94,7 +94,7 @@ class ProcessSpiderExceptionReRaise(SpiderMiddlewareTestCase): class RaiseExceptionProcessSpiderOutputMiddleware: def process_spider_output(self, response, result, spider): - 1/0 + 1 / 0 self.mwman._add_middleware(ProcessSpiderExceptionReturnNoneMiddleware()) self.mwman._add_middleware(RaiseExceptionProcessSpiderOutputMiddleware()) diff --git a/tests/test_spidermiddleware_output_chain.py b/tests/test_spidermiddleware_output_chain.py index b19a74609..b26353d6c 100644 --- a/tests/test_spidermiddleware_output_chain.py +++ b/tests/test_spidermiddleware_output_chain.py @@ -125,7 +125,7 @@ class NotGeneratorCallbackSpider(Spider): yield Request(self.mockserver.url('/status?n=200')) def parse(self, response): - return [{'test': 1}, {'test': 1/0}] + return [{'test': 1}, {'test': 1 / 0}] # ================================================================================ diff --git a/tests/test_utils_defer.py b/tests/test_utils_defer.py index 89b5fb4fb..a3b6e64f1 100644 --- a/tests/test_utils_defer.py +++ b/tests/test_utils_defer.py @@ -104,7 +104,7 @@ class IterErrbackTest(unittest.TestCase): def iterbad(): for x in range(10): if x == 5: - a = 1/0 + a = 1 / 0 yield x errors = [] diff --git a/tests/test_utils_log.py b/tests/test_utils_log.py index 2c23f3616..21100aeb8 100644 --- a/tests/test_utils_log.py +++ b/tests/test_utils_log.py @@ -16,7 +16,7 @@ class FailureToExcInfoTest(unittest.TestCase): def test_failure(self): try: - 0/0 + 0 / 0 except ZeroDivisionError: exc_info = sys.exc_info() failure = Failure() diff --git a/tests/test_utils_signal.py b/tests/test_utils_signal.py index e5f6f0ed4..9f6da09ed 100644 --- a/tests/test_utils_signal.py +++ b/tests/test_utils_signal.py @@ -44,7 +44,7 @@ class SendCatchLogTest(unittest.TestCase): def error_handler(self, arg, handlers_called): handlers_called.add(self.error_handler) - a = 1/0 + a = 1 / 0 def ok_handler(self, arg, handlers_called): handlers_called.add(self.ok_handler) diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 9f1acbc75..1e18494c3 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -30,7 +30,7 @@ class UrlUtilsTest(unittest.TestCase): url = 'javascript:%20document.orderform_2581_1190810811.mode.value=%27add%27;%20javascript:%20document.orderform_2581_1190810811.submit%28%29' self.assertFalse(url_is_from_any_domain(url, ['testdomain.com'])) - self.assertFalse(url_is_from_any_domain(url+'.testdomain.com', ['testdomain.com'])) + self.assertFalse(url_is_from_any_domain(url + '.testdomain.com', ['testdomain.com'])) def test_url_is_from_spider(self): spider = Spider(name='example.com') diff --git a/tests/test_webclient.py b/tests/test_webclient.py index b602a3ea0..99a998a46 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -51,22 +51,22 @@ class ParseUrlTestCase(unittest.TestCase): ("http://127.0.0.1?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), ("http://127.0.0.1/?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), ("http://127.0.0.1/foo?c=v&c2=v2#frag", ('http', lip, lip, 80, '/foo?c=v&c2=v2')), - ("http://127.0.0.1:100?c=v&c2=v2#fragment", ('http', lip+':100', lip, 100, '/?c=v&c2=v2')), - ("http://127.0.0.1:100/?c=v&c2=v2#frag", ('http', lip+':100', lip, 100, '/?c=v&c2=v2')), - ("http://127.0.0.1:100/foo?c=v&c2=v2#frag", ('http', lip+':100', lip, 100, '/foo?c=v&c2=v2')), + ("http://127.0.0.1:100?c=v&c2=v2#fragment", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), + ("http://127.0.0.1:100/?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), + ("http://127.0.0.1:100/foo?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/foo?c=v&c2=v2')), ("http://127.0.0.1", ('http', lip, lip, 80, '/')), ("http://127.0.0.1/", ('http', lip, lip, 80, '/')), ("http://127.0.0.1/foo", ('http', lip, lip, 80, '/foo')), ("http://127.0.0.1?param=value", ('http', lip, lip, 80, '/?param=value')), ("http://127.0.0.1/?param=value", ('http', lip, lip, 80, '/?param=value')), - ("http://127.0.0.1:12345/foo", ('http', lip+':12345', lip, 12345, '/foo')), + ("http://127.0.0.1:12345/foo", ('http', lip + ':12345', lip, 12345, '/foo')), ("http://spam:12345/foo", ('http', 'spam:12345', 'spam', 12345, '/foo')), ("http://spam.test.org/foo", ('http', 'spam.test.org', 'spam.test.org', 80, '/foo')), ("https://127.0.0.1/foo", ('https', lip, lip, 443, '/foo')), ("https://127.0.0.1/?param=value", ('https', lip, lip, 443, '/?param=value')), - ("https://127.0.0.1:12345/", ('https', lip+':12345', lip, 12345, '/')), + ("https://127.0.0.1:12345/", ('https', lip + ':12345', lip, 12345, '/')), ("http://scrapytest.org/foo ", ('http', 'scrapytest.org', 'scrapytest.org', 80, '/foo')), ("http://egg:7890 ", ('http', 'egg:7890', 'egg', 7890, '/')), From 03ed9e17867b8c7533d08ef28108a67305050e9a Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Fri, 21 Feb 2020 09:29:29 +0100 Subject: [PATCH 221/313] delete old deprecated functions from scrapy.utils.python --- scrapy/utils/python.py | 44 ------------------------------------------ 1 file changed, 44 deletions(-) diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index e5582cc18..e95a4648e 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -4,7 +4,6 @@ This module contains essential stuff that should've come with Python itself ;) import errno import gc import inspect -import os import re import sys import weakref @@ -165,14 +164,6 @@ _BINARYCHARS = {to_bytes(chr(i)) for i in range(32)} - {b"\0", b"\t", b"\n", b"\ _BINARYCHARS |= {ord(ch) for ch in _BINARYCHARS} -@deprecated("scrapy.utils.python.binary_is_text") -def isbinarytext(text): - """ This function is deprecated. - Please use scrapy.utils.python.binary_is_text, which was created to be more - clear about the functions behavior: it is behaving inverted to this one. """ - return not binary_is_text(text) - - def binary_is_text(data): """ Returns ``True`` if the given ``data`` argument (a ``bytes`` object) does not contain unprintable control characters. @@ -293,41 +284,6 @@ class WeakKeyCache(object): return self._weakdict[key] -@deprecated -def stringify_dict(dct_or_tuples, encoding='utf-8', keys_only=True): - """Return a (new) dict with unicode keys (and values when "keys_only" is - False) of the given dict converted to strings. ``dct_or_tuples`` can be a - dict or a list of tuples, like any dict ``__init__`` method supports. - """ - d = {} - for k, v in dict(dct_or_tuples).items(): - k = k.encode(encoding) if isinstance(k, str) else k - if not keys_only: - v = v.encode(encoding) if isinstance(v, str) else v - d[k] = v - return d - - -@deprecated -def is_writable(path): - """Return True if the given path can be written (if it exists) or created - (if it doesn't exist) - """ - if os.path.exists(path): - return os.access(path, os.W_OK) - else: - return os.access(os.path.dirname(path), os.W_OK) - - -@deprecated -def setattr_default(obj, name, value): - """Set attribute value, but only if it's not already set. Similar to - setdefault() for dicts. - """ - if not hasattr(obj, name): - setattr(obj, name, value) - - def retry_on_eintr(function, *args, **kw): """Run a function and retry it while getting EINTR errors""" while True: From b49ece0b8781c1d53cdec77b96445826a089afc1 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Fri, 21 Feb 2020 08:58:32 +0100 Subject: [PATCH 222/313] fix E701 and E271 flake8 --- pytest.ini | 14 +++++++------- scrapy/utils/iterators.py | 6 ++++-- tests/test_item.py | 21 ++++++++++++++------- tests/test_pipeline_files.py | 2 +- tests/test_squeues.py | 4 +++- tests/test_utils_python.py | 4 +++- 6 files changed, 32 insertions(+), 19 deletions(-) diff --git a/pytest.ini b/pytest.ini index 7806620d5..acdb5a27a 100644 --- a/pytest.ini +++ b/pytest.ini @@ -128,7 +128,7 @@ flake8-ignore = scrapy/utils/gz.py E501 W504 scrapy/utils/http.py F403 scrapy/utils/httpobj.py E501 - scrapy/utils/iterators.py E501 E701 + scrapy/utils/iterators.py E501 scrapy/utils/log.py E128 E501 scrapy/utils/markup.py F403 scrapy/utils/misc.py E501 @@ -141,7 +141,7 @@ flake8-ignore = scrapy/utils/response.py E501 E128 scrapy/utils/signal.py E501 E128 scrapy/utils/sitemap.py E501 - scrapy/utils/spider.py E271 E501 + scrapy/utils/spider.py E501 scrapy/utils/ssl.py E501 scrapy/utils/test.py E501 scrapy/utils/url.py E501 F403 E128 F405 @@ -181,7 +181,7 @@ flake8-ignore = tests/test_crawl.py E501 E741 E265 tests/test_crawler.py F841 E501 tests/test_dependencies.py F841 E501 - tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E701 E126 E123 + tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126 @@ -204,7 +204,7 @@ flake8-ignore = tests/test_http_headers.py E501 tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 tests/test_http_response.py E501 E128 E265 - tests/test_item.py E701 E128 F841 + tests/test_item.py E128 F841 tests/test_link.py E501 tests/test_linkextractors.py E501 E128 E124 tests/test_loader.py E501 E731 E741 E128 E117 E241 @@ -212,7 +212,7 @@ flake8-ignore = tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 tests/test_pipeline_crawl.py E131 E501 E128 E126 - tests/test_pipeline_files.py E501 E272 + tests/test_pipeline_files.py E501 tests/test_pipeline_images.py F841 E501 tests/test_pipeline_media.py E501 E741 E731 E128 E502 tests/test_proxy_connect.py E501 E741 @@ -227,7 +227,7 @@ flake8-ignore = tests/test_spidermiddleware_offsite.py E501 E128 E111 tests/test_spidermiddleware_output_chain.py E501 tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 - tests/test_squeues.py E501 E701 E741 + tests/test_squeues.py E501 E741 tests/test_utils_asyncio.py E501 tests/test_utils_conf.py E501 E128 tests/test_utils_curl.py E501 @@ -237,7 +237,7 @@ flake8-ignore = tests/test_utils_http.py E501 E128 W504 tests/test_utils_iterators.py E501 E128 E129 E241 tests/test_utils_log.py E741 - tests/test_utils_python.py E501 E731 E701 + tests/test_utils_python.py E501 E731 tests/test_utils_reqser.py E501 E128 tests/test_utils_request.py E501 E128 tests/test_utils_response.py E501 diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index 3c0cb68c3..7849174fb 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -101,8 +101,10 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): lines = StringIO(_body_or_str(obj, unicode=True)) kwargs = {} - if delimiter: kwargs["delimiter"] = delimiter - if quotechar: kwargs["quotechar"] = quotechar + if delimiter: + kwargs["delimiter"] = delimiter + if quotechar: + kwargs["quotechar"] = quotechar csv_r = csv.reader(lines, **kwargs) if not headers: diff --git a/tests/test_item.py b/tests/test_item.py index 823bf1ced..f70632d57 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -149,13 +149,15 @@ class ItemTest(unittest.TestCase): fields = {'load': Field(default='A')} save = Field(default='A') - class B(A): pass + class B(A): + pass class C(Item): fields = {'load': Field(default='C')} save = Field(default='C') - class D(B, C): pass + class D(B, C): + pass item = D(save='X', load='Y') self.assertEqual(item['save'], 'X') @@ -164,7 +166,8 @@ class ItemTest(unittest.TestCase): 'save': {'default': 'A'}}) # D class inverted - class E(C, B): pass + class E(C, B): + pass self.assertEqual(E(save='X')['save'], 'X') self.assertEqual(E(load='X')['load'], 'X') @@ -177,7 +180,8 @@ class ItemTest(unittest.TestCase): save = Field(default='A') load = Field(default='A') - class B(A): pass + class B(A): + pass class C(A): fields = {'update': Field(default='C')} @@ -206,14 +210,16 @@ class ItemTest(unittest.TestCase): fields = {'load': Field(default='A')} save = Field(default='A') - class B(A): pass + class B(A): + pass class C(object): fields = {'load': Field(default='C')} not_allowed = Field(default='not_allowed') save = Field(default='C') - class D(B, C): pass + class D(B, C): + pass self.assertRaises(KeyError, D, not_allowed='value') self.assertEqual(D(save='X')['save'], 'X') @@ -221,7 +227,8 @@ class ItemTest(unittest.TestCase): 'load': {'default': 'A'}}) # D class inverted - class E(C, B): pass + class E(C, B): + pass self.assertRaises(KeyError, E, not_allowed='value') self.assertEqual(E(save='X')['save'], 'X') diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index 799782647..f155db4ce 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -272,7 +272,7 @@ class FilesPipelineTestCaseCustomSettings(unittest.TestCase): prefix = pipeline_cls.__name__.upper() settings = self._generate_fake_settings(prefix=prefix) user_pipeline = pipeline_cls.from_settings(Settings(settings)) - for pipe_cls_attr, settings_attr, pipe_inst_attr in self.file_cls_attr_settings_map: + for pipe_cls_attr, settings_attr, pipe_inst_attr in self.file_cls_attr_settings_map: custom_value = settings.get(prefix + "_" + settings_attr) self.assertNotEqual(custom_value, self.default_cls_settings[pipe_cls_attr]) self.assertEqual(getattr(user_pipeline, pipe_inst_attr), custom_value) diff --git a/tests/test_squeues.py b/tests/test_squeues.py index d5fcf2f7f..f6970162e 100644 --- a/tests/test_squeues.py +++ b/tests/test_squeues.py @@ -31,7 +31,9 @@ def nonserializable_object_test(self): self.assertRaises(ValueError, q.push, lambda x: x) else: # Use a different unpickleable object - class A(object): pass + class A(object): + pass + a = A() a.__reduce__ = a.__reduce_ex__ = None self.assertRaises(ValueError, q.push, a) diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index 4202e8c89..ec5b4c596 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -153,7 +153,9 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertFalse(equal_attributes(a, b, [compare_z, 'x'])) def test_weakkeycache(self): - class _Weakme(object): pass + class _Weakme(object): + pass + _values = count() wk = WeakKeyCache(lambda k: next(_values)) k = _Weakme() From 9ad10bb6f727a3f1c5c59d490f444ebb32de97c6 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Fri, 21 Feb 2020 09:05:42 +0100 Subject: [PATCH 223/313] fix E131 --- pytest.ini | 2 +- tests/test_pipeline_crawl.py | 12 ++++++------ 2 files changed, 7 insertions(+), 7 deletions(-) diff --git a/pytest.ini b/pytest.ini index acdb5a27a..2120264e0 100644 --- a/pytest.ini +++ b/pytest.ini @@ -211,7 +211,7 @@ flake8-ignore = tests/test_logformatter.py E128 E501 E122 tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 - tests/test_pipeline_crawl.py E131 E501 E128 E126 + tests/test_pipeline_crawl.py E501 E128 E126 tests/test_pipeline_files.py E501 tests/test_pipeline_images.py F841 E501 tests/test_pipeline_media.py E501 E741 E731 E128 E502 diff --git a/tests/test_pipeline_crawl.py b/tests/test_pipeline_crawl.py index fb72c9d6d..962c33144 100644 --- a/tests/test_pipeline_crawl.py +++ b/tests/test_pipeline_crawl.py @@ -26,10 +26,9 @@ class MediaDownloadSpider(SimpleSpider): self.media_key: [], self.media_urls_key: [ self._process_url(response.urljoin(href)) - for href in response.xpath(''' - //table[thead/tr/th="Filename"] - /tbody//a/@href - ''').getall()], + for href in response.xpath( + '//table[thead/tr/th="Filename"]/tbody//a/@href' + ).getall()], } yield item @@ -99,8 +98,9 @@ class FileDownloadCrawlTestCase(TestCase): if self.expected_checksums is not None: checksums = set( i['checksum'] - for item in items - for i in item[self.media_key]) + for item in items + for i in item[self.media_key] + ) self.assertEqual(checksums, self.expected_checksums) # check that the image files where actually written to the media store From 69a8648bef6df38a5b7e79f9fbecb98869416654 Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Fri, 21 Feb 2020 09:13:28 +0100 Subject: [PATCH 224/313] fix E251 --- pytest.ini | 4 ++-- tests/test_downloadermiddleware_httpcompression.py | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/pytest.ini b/pytest.ini index 2120264e0..58f1cfeb3 100644 --- a/pytest.ini +++ b/pytest.ini @@ -189,10 +189,10 @@ flake8-ignore = tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 tests/test_downloadermiddleware_httpcache.py E501 - tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123 + tests/test_downloadermiddleware_httpcompression.py E501 E126 E123 tests/test_downloadermiddleware_httpproxy.py E501 E128 tests/test_downloadermiddleware_redirect.py E501 E128 E127 - tests/test_downloadermiddleware_retry.py E501 E128 E251 E126 + tests/test_downloadermiddleware_retry.py E501 E128 E126 tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 tests/test_dupefilters.py E501 E741 E128 E124 diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index 64488841a..106ca3360 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -245,7 +245,7 @@ class HttpCompressionTest(TestCase): response.headers['Content-Type'] = 'application/gzip' request = response.request request.method = 'HEAD' - response = response.replace(body = None) + response = response.replace(body=None) newresponse = self.mw.process_response(request, response, self.spider) self.assertIs(newresponse, response) self.assertEqual(response.body, b'') From 6e8e117aee4ddc5d6f6970019be212198d0b9e7a Mon Sep 17 00:00:00 2001 From: Marc Hernandez Cabot <noviluni@gmail.com> Date: Fri, 21 Feb 2020 09:14:55 +0100 Subject: [PATCH 225/313] fix flake E211 --- pytest.ini | 2 +- tests/test_utils_url.py | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/pytest.ini b/pytest.ini index 58f1cfeb3..141a13a4f 100644 --- a/pytest.ini +++ b/pytest.ini @@ -243,7 +243,7 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 E731 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_url.py E501 E127 E211 E125 E501 E241 E126 E123 + tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123 tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 1e18494c3..7abff8281 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -202,7 +202,7 @@ def create_skipped_scheme_t(args): return do_expected -for k, args in enumerate ([ +for k, args in enumerate([ ('/index', 'file://'), ('/index.html', 'file://'), ('./index.html', 'file://'), @@ -230,7 +230,7 @@ for k, args in enumerate ([ ], start=1): t_method = create_guess_scheme_t(args) t_method.__name__ = 'test_uri_%03d' % k - setattr (GuessSchemeTest, t_method.__name__, t_method) + setattr(GuessSchemeTest, t_method.__name__, t_method) # TODO: the following tests do not pass with current implementation for k, args in enumerate([ @@ -239,7 +239,7 @@ for k, args in enumerate([ ], start=1): t_method = create_skipped_scheme_t(args) t_method.__name__ = 'test_uri_skipped_%03d' % k - setattr (GuessSchemeTest, t_method.__name__, t_method) + setattr(GuessSchemeTest, t_method.__name__, t_method) class StripUrl(unittest.TestCase): From 67ee0b097fe15aefa787bce64f6fa085d38e69d8 Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Sat, 22 Feb 2020 17:02:57 +0500 Subject: [PATCH 226/313] Remove specifics of downstream request queues from scheduler (#3884) * move serialization/deserialization logic to downstream queues * make memory queues conform to common interface * make ScrapyPriorityQueue conform common interface * ScrapyPriorityQueue works with disk * make key as string * return list instead of dict as earlier * downloader aware pq works with new interface * we don`t need these methods anymore * create directories for files * remove dummy priority * remove priority as parameter, let every queue decide for itself * rename obj to request * DownloaderAwarePriorityQueue is too thin wrapper around _SlotPriorityQueues, just remove second one * remove priority as parameter, let every queue decide for itself * rename argument * more granular class separation * python2 compatible * one more argument for common interface * more simple downstream queue interface * single place for easier customization * rename function * shorter * shorter * use named arguments * fix typo * add docstring * Update scrapy/pqueues.py Co-Authored-By: Mikhail Korobov <kmike84@gmail.com> * Update scrapy/pqueues.py Co-Authored-By: Mikhail Korobov <kmike84@gmail.com> * 4 spaces indentation * we ok with existing directories * remove unused import * rename method * remove unused imports * it has no sense now * relining * note about queues * add value * Revert "it has no sense now" This reverts commit b61604275ba090ebd8e30a6d3a6fbe281c74c189. * pep8 E261 * pep8 E303 * pep8 E501 * pep8 E123 * pep8 E123 * use create instance * remove excessive import Co-authored-by: Mikhail Korobov <kmike84@gmail.com> --- scrapy/core/scheduler.py | 30 +++--- scrapy/pqueues.py | 201 +++++++++++++++++++++------------------ scrapy/squeues.py | 100 +++++++++++++++++-- tests/test_squeues.py | 7 +- 4 files changed, 214 insertions(+), 124 deletions(-) diff --git a/scrapy/core/scheduler.py b/scrapy/core/scheduler.py index 975aede0c..e184ed50e 100644 --- a/scrapy/core/scheduler.py +++ b/scrapy/core/scheduler.py @@ -119,7 +119,7 @@ class Scheduler(object): if self.dqs is None: return try: - self.dqs.push(request, -request.priority) + self.dqs.push(request) except ValueError as e: # non serializable request if self.logunser: msg = ("Unable to serialize request: %(request)s - reason:" @@ -135,35 +135,29 @@ class Scheduler(object): return True def _mqpush(self, request): - self.mqs.push(request, -request.priority) + self.mqs.push(request) def _dqpop(self): if self.dqs: return self.dqs.pop() - def _newmq(self, priority): - """ Factory for creating memory queues. """ - return self.mqclass() - - def _newdq(self, priority): - """ Factory for creating disk queues. """ - path = join(self.dqdir, 'p%s' % (priority, )) - return self.dqclass(path) - def _mq(self): """ Create a new priority queue instance, with in-memory storage """ - return create_instance(self.pqclass, None, self.crawler, self._newmq, - serialize=False) + return create_instance(self.pqclass, + settings=None, + crawler=self.crawler, + downstream_queue_cls=self.mqclass, + key='') def _dq(self): """ Create a new priority queue instance, with disk storage """ state = self._read_dqs_state(self.dqdir) q = create_instance(self.pqclass, - None, - self.crawler, - self._newdq, - state, - serialize=True) + settings=None, + crawler=self.crawler, + downstream_queue_cls=self.dqclass, + key=self.dqdir, + startprios=state) if q: logger.info("Resuming crawl (%(queuesize)d requests scheduled)", {'queuesize': len(q)}, extra={'spider': self.spider}) diff --git a/scrapy/pqueues.py b/scrapy/pqueues.py index 717ed4d27..1afe58dab 100644 --- a/scrapy/pqueues.py +++ b/scrapy/pqueues.py @@ -1,11 +1,7 @@ import hashlib import logging -from collections import namedtuple - -from queuelib import PriorityQueue - -from scrapy.utils.reqser import request_to_dict, request_from_dict +from scrapy.utils.misc import create_instance logger = logging.getLogger(__name__) @@ -29,88 +25,89 @@ def _path_safe(text): return '-'.join([pathable_slot, unique_slot]) -class _Priority(namedtuple("_Priority", ["priority", "slot"])): - """ Slot-specific priority. It is a hack - ``(priority, slot)`` tuple - which can be used instead of int priorities in queues: +class ScrapyPriorityQueue: + """A priority queue implemented using multiple internal queues (typically, + FIFO queues). It uses one internal queue for each priority value. The internal + queue must implement the following methods: + + * push(obj) + * pop() + * close() + * __len__() + + ``__init__`` method of ScrapyPriorityQueue receives a downstream_queue_cls + argument, which is a class used to instantiate a new (internal) queue when + a new priority is allocated. + + Only integer priorities should be used. Lower numbers are higher + priorities. + + startprios is a sequence of priorities to start with. If the queue was + previously closed leaving some priority buckets non-empty, those priorities + should be passed in startprios. - * they are ordered in the same way - order is still by priority value, - min(prios) works; - * str(p) representation is guaranteed to be different when slots - are different - this is important because str(p) is used to create - queue files on disk; - * they have readable str(p) representation which is safe - to use as a file name. """ - __slots__ = () - def __str__(self): - return '%s_%s' % (self.priority, _path_safe(str(self.slot))) + @classmethod + def from_crawler(cls, crawler, downstream_queue_cls, key, startprios=()): + return cls(crawler, downstream_queue_cls, key, startprios) + def __init__(self, crawler, downstream_queue_cls, key, startprios=()): + self.crawler = crawler + self.downstream_queue_cls = downstream_queue_cls + self.key = key + self.queues = {} + self.curprio = None + self.init_prios(startprios) -class _SlotPriorityQueues(object): - """ Container for multiple priority queues. """ - def __init__(self, pqfactory, slot_startprios=None): - """ - ``pqfactory`` is a factory for creating new PriorityQueues. - It must be a function which accepts a single optional ``startprios`` - argument, with a list of priorities to create queues for. + def init_prios(self, startprios): + if not startprios: + return - ``slot_startprios`` is a ``{slot: startprios}`` dict. - """ - self.pqfactory = pqfactory - self.pqueues = {} # slot -> priority queue - for slot, startprios in (slot_startprios or {}).items(): - self.pqueues[slot] = self.pqfactory(startprios) + for priority in startprios: + self.queues[priority] = self.qfactory(priority) - def pop_slot(self, slot): - """ Pop an object from a priority queue for this slot """ - queue = self.pqueues[slot] - request = queue.pop() - if len(queue) == 0: - del self.pqueues[slot] - return request + self.curprio = min(startprios) - def push_slot(self, slot, obj, priority): - """ Push an object to a priority queue for this slot """ - if slot not in self.pqueues: - self.pqueues[slot] = self.pqfactory() - queue = self.pqueues[slot] - queue.push(obj, priority) + def qfactory(self, key): + return create_instance(self.downstream_queue_cls, + None, + self.crawler, + self.key + '/' + str(key)) + + def priority(self, request): + return -request.priority + + def push(self, request): + priority = self.priority(request) + if priority not in self.queues: + self.queues[priority] = self.qfactory(priority) + q = self.queues[priority] + q.push(request) # this may fail (eg. serialization error) + if self.curprio is None or priority < self.curprio: + self.curprio = priority + + def pop(self): + if self.curprio is None: + return + q = self.queues[self.curprio] + m = q.pop() + if not q: + del self.queues[self.curprio] + q.close() + prios = [p for p, q in self.queues.items() if q] + self.curprio = min(prios) if prios else None + return m def close(self): - active = {slot: queue.close() - for slot, queue in self.pqueues.items()} - self.pqueues.clear() + active = [] + for p, q in self.queues.items(): + active.append(p) + q.close() return active def __len__(self): - return sum(len(x) for x in self.pqueues.values()) if self.pqueues else 0 - - -class ScrapyPriorityQueue(PriorityQueue): - """ - PriorityQueue which works with scrapy.Request instances and - can optionally convert them to/from dicts before/after putting to a queue. - """ - def __init__(self, crawler, qfactory, startprios=(), serialize=False): - super(ScrapyPriorityQueue, self).__init__(qfactory, startprios) - self.serialize = serialize - self.spider = crawler.spider - - @classmethod - def from_crawler(cls, crawler, qfactory, startprios=(), serialize=False): - return cls(crawler, qfactory, startprios, serialize) - - def push(self, request, priority=0): - if self.serialize: - request = request_to_dict(request, self.spider) - super(ScrapyPriorityQueue, self).push(request, priority) - - def pop(self): - request = super(ScrapyPriorityQueue, self).pop() - if request and self.serialize: - request = request_from_dict(request, self.spider) - return request + return sum(len(x) for x in self.queues.values()) if self.queues else 0 class DownloaderInterface(object): @@ -133,16 +130,16 @@ class DownloaderInterface(object): class DownloaderAwarePriorityQueue(object): - """ PriorityQueue which takes Downlaoder activity in account: + """ PriorityQueue which takes Downloader activity in account: domains (slots) with the least amount of active downloads are dequeued first. """ @classmethod - def from_crawler(cls, crawler, qfactory, slot_startprios=None, serialize=False): - return cls(crawler, qfactory, slot_startprios, serialize) + def from_crawler(cls, crawler, downstream_queue_cls, key, startprios=()): + return cls(crawler, downstream_queue_cls, key, startprios) - def __init__(self, crawler, qfactory, slot_startprios=None, serialize=False): + def __init__(self, crawler, downstream_queue_cls, key, slot_startprios=()): if crawler.settings.getint('CONCURRENT_REQUESTS_PER_IP') != 0: raise ValueError('"%s" does not support CONCURRENT_REQUESTS_PER_IP' % (self.__class__,)) @@ -156,35 +153,49 @@ class DownloaderAwarePriorityQueue(object): "queue class can be resumed." % slot_startprios.__class__) - slot_startprios = { - slot: [_Priority(p, slot) for p in startprios] - for slot, startprios in (slot_startprios or {}).items()} - - def pqfactory(startprios=()): - return ScrapyPriorityQueue(crawler, qfactory, startprios, serialize) - self._slot_pqueues = _SlotPriorityQueues(pqfactory, slot_startprios) - self.serialize = serialize self._downloader_interface = DownloaderInterface(crawler) + self.downstream_queue_cls = downstream_queue_cls + self.key = key + self.crawler = crawler + + self.pqueues = {} # slot -> priority queue + for slot, startprios in (slot_startprios or {}).items(): + self.pqueues[slot] = self.pqfactory(slot, startprios) + + def pqfactory(self, slot, startprios=()): + return ScrapyPriorityQueue(self.crawler, + self.downstream_queue_cls, + self.key + '/' + _path_safe(slot), + startprios) def pop(self): - stats = self._downloader_interface.stats(self._slot_pqueues.pqueues) + stats = self._downloader_interface.stats(self.pqueues) if not stats: return slot = min(stats)[1] - request = self._slot_pqueues.pop_slot(slot) + queue = self.pqueues[slot] + request = queue.pop() + if len(queue) == 0: + del self.pqueues[slot] return request - def push(self, request, priority): + def push(self, request): slot = self._downloader_interface.get_slot_key(request) - priority_slot = _Priority(priority=priority, slot=slot) - self._slot_pqueues.push_slot(slot, request, priority_slot) + if slot not in self.pqueues: + self.pqueues[slot] = self.pqfactory(slot) + queue = self.pqueues[slot] + queue.push(request) def close(self): - active = self._slot_pqueues.close() - return {slot: [p.priority for p in startprios] - for slot, startprios in active.items()} + active = {slot: queue.close() + for slot, queue in self.pqueues.items()} + self.pqueues.clear() + return active def __len__(self): - return len(self._slot_pqueues) + return sum(len(x) for x in self.pqueues.values()) if self.pqueues else 0 + + def __contains__(self, slot): + return slot in self.pqueues diff --git a/scrapy/squeues.py b/scrapy/squeues.py index d5d3be67e..d0686dac3 100644 --- a/scrapy/squeues.py +++ b/scrapy/squeues.py @@ -3,10 +3,27 @@ Scheduler queues """ import marshal +import os import pickle from queuelib import queue +from scrapy.utils.reqser import request_to_dict, request_from_dict + + +def _with_mkdir(queue_class): + + class DirectoriesCreated(queue_class): + + def __init__(self, path, *args, **kwargs): + dirname = os.path.dirname(path) + if not os.path.exists(dirname): + os.makedirs(dirname, exist_ok=True) + + super(DirectoriesCreated, self).__init__(path, *args, **kwargs) + + return DirectoriesCreated + def _serializable_queue(queue_class, serialize, deserialize): @@ -24,6 +41,44 @@ def _serializable_queue(queue_class, serialize, deserialize): return SerializableQueue +def _scrapy_serialization_queue(queue_class): + + class ScrapyRequestQueue(queue_class): + + def __init__(self, crawler, key): + self.spider = crawler.spider + super(ScrapyRequestQueue, self).__init__(key) + + @classmethod + def from_crawler(cls, crawler, key, *args, **kwargs): + return cls(crawler, key) + + def push(self, request): + request = request_to_dict(request, self.spider) + return super(ScrapyRequestQueue, self).push(request) + + def pop(self): + request = super(ScrapyRequestQueue, self).pop() + + if not request: + return None + + request = request_from_dict(request, self.spider) + return request + + return ScrapyRequestQueue + + +def _scrapy_non_serialization_queue(queue_class): + + class ScrapyRequestQueue(queue_class): + @classmethod + def from_crawler(cls, crawler, *args, **kwargs): + return cls() + + return ScrapyRequestQueue + + def _pickle_serialize(obj): try: return pickle.dumps(obj, protocol=2) @@ -34,13 +89,38 @@ def _pickle_serialize(obj): raise ValueError(str(e)) -PickleFifoDiskQueue = _serializable_queue(queue.FifoDiskQueue, - _pickle_serialize, pickle.loads) -PickleLifoDiskQueue = _serializable_queue(queue.LifoDiskQueue, - _pickle_serialize, pickle.loads) -MarshalFifoDiskQueue = _serializable_queue(queue.FifoDiskQueue, - marshal.dumps, marshal.loads) -MarshalLifoDiskQueue = _serializable_queue(queue.LifoDiskQueue, - marshal.dumps, marshal.loads) -FifoMemoryQueue = queue.FifoMemoryQueue -LifoMemoryQueue = queue.LifoMemoryQueue +PickleFifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.FifoDiskQueue), + _pickle_serialize, + pickle.loads +) +PickleLifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.LifoDiskQueue), + _pickle_serialize, + pickle.loads +) +MarshalFifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.FifoDiskQueue), + marshal.dumps, + marshal.loads +) +MarshalLifoDiskQueueNonRequest = _serializable_queue( + _with_mkdir(queue.LifoDiskQueue), + marshal.dumps, + marshal.loads +) + +PickleFifoDiskQueue = _scrapy_serialization_queue( + PickleFifoDiskQueueNonRequest +) +PickleLifoDiskQueue = _scrapy_serialization_queue( + PickleLifoDiskQueueNonRequest +) +MarshalFifoDiskQueue = _scrapy_serialization_queue( + MarshalFifoDiskQueueNonRequest +) +MarshalLifoDiskQueue = _scrapy_serialization_queue( + MarshalLifoDiskQueueNonRequest +) +FifoMemoryQueue = _scrapy_non_serialization_queue(queue.FifoMemoryQueue) +LifoMemoryQueue = _scrapy_non_serialization_queue(queue.LifoMemoryQueue) diff --git a/tests/test_squeues.py b/tests/test_squeues.py index d5fcf2f7f..5c626fbcb 100644 --- a/tests/test_squeues.py +++ b/tests/test_squeues.py @@ -1,7 +1,12 @@ import pickle from queuelib.tests import test_queue as t -from scrapy.squeues import MarshalFifoDiskQueue, MarshalLifoDiskQueue, PickleFifoDiskQueue, PickleLifoDiskQueue +from scrapy.squeues import ( + MarshalFifoDiskQueueNonRequest as MarshalFifoDiskQueue, + MarshalLifoDiskQueueNonRequest as MarshalLifoDiskQueue, + PickleFifoDiskQueueNonRequest as PickleFifoDiskQueue, + PickleLifoDiskQueueNonRequest as PickleLifoDiskQueue +) from scrapy.item import Item, Field from scrapy.http import Request from scrapy.loader import ItemLoader From 9d983c1b9962a018686111e20e42d25bbffb579e Mon Sep 17 00:00:00 2001 From: elacuesta <elacuesta@users.noreply.github.com> Date: Sat, 22 Feb 2020 09:20:31 -0300 Subject: [PATCH 227/313] Expose certificate for HTTPS responses (#4054) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * Expose certificate for HTTPS responses * Fix test (missing inlineCallbacks decorator) * Note about Response.certificate * Explicitly cover None as the default value of Response.certificate Co-authored-by: Adrián Chaves <adrian@chaves.io> --- docs/topics/request-response.rst | 12 +++++++++- scrapy/core/downloader/handlers/http11.py | 22 +++++++++++------ scrapy/http/response/__init__.py | 5 ++-- tests/test_crawl.py | 29 +++++++++++++++++++++++ 4 files changed, 58 insertions(+), 10 deletions(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 672c0b3d6..f009facd6 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -609,7 +609,10 @@ Response objects :param request: the initial value of the :attr:`Response.request` attribute. This represents the :class:`Request` that generated this response. - :type request: :class:`Request` object + :type request: scrapy.http.Request + + :param certificate: an object representing the server's SSL certificate. + :type certificate: twisted.internet.ssl.Certificate .. attribute:: Response.url @@ -691,6 +694,13 @@ Response objects they're shown on the string representation of the Response (`__str__` method) which is used by the engine for logging. + .. attribute:: Response.certificate + + A :class:`twisted.internet.ssl.Certificate` object representing + the server's SSL certificate. + + Only populated for ``https`` responses, ``None`` otherwise. + .. method:: Response.copy() Returns a new Response which is a copy of this Response. diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 5a5f6cf0a..93951d3b5 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -3,11 +3,12 @@ import logging import re import warnings +from contextlib import suppress from io import BytesIO from time import time from urllib.parse import urldefrag -from twisted.internet import defer, protocol, reactor +from twisted.internet import defer, protocol, reactor, ssl from twisted.internet.endpoints import TCP4ClientEndpoint from twisted.internet.error import TimeoutError from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI @@ -382,7 +383,7 @@ class ScrapyAgent(object): def _cb_bodyready(self, txresponse, request): # deliverBody hangs for responses without body if txresponse.length == 0: - return txresponse, b'', None + return txresponse, b'', None, None maxsize = request.meta.get('download_maxsize', self._maxsize) warnsize = request.meta.get('download_warnsize', self._warnsize) @@ -418,11 +419,12 @@ class ScrapyAgent(object): return d def _cb_bodydone(self, result, request, url): - txresponse, body, flags = result + txresponse, body, flags, certificate = result status = int(txresponse.code) headers = Headers(txresponse.headers.getAllRawHeaders()) respcls = responsetypes.from_args(headers=headers, url=url, body=body) - return respcls(url=url, status=status, headers=headers, body=body, flags=flags) + return respcls(url=url, status=status, headers=headers, body=body, + flags=flags, certificate=certificate) @implementer(IBodyProducer) @@ -456,6 +458,12 @@ class _ResponseReader(protocol.Protocol): self._fail_on_dataloss_warned = False self._reached_warnsize = False self._bytes_received = 0 + self._certificate = None + + def connectionMade(self): + if self._certificate is None: + with suppress(AttributeError): + self._certificate = ssl.Certificate(self.transport._producer.getPeerCertificate()) def dataReceived(self, bodyBytes): # This maybe called several times after cancel was called with buffered data. @@ -488,16 +496,16 @@ class _ResponseReader(protocol.Protocol): body = self._bodybuf.getvalue() if reason.check(ResponseDone): - self._finished.callback((self._txresponse, body, None)) + self._finished.callback((self._txresponse, body, None, self._certificate)) return if reason.check(PotentialDataLoss): - self._finished.callback((self._txresponse, body, ['partial'])) + self._finished.callback((self._txresponse, body, ['partial'], self._certificate)) return if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons): if not self._fail_on_dataloss: - self._finished.callback((self._txresponse, body, ['dataloss'])) + self._finished.callback((self._txresponse, body, ['dataloss'], self._certificate)) return elif not self._fail_on_dataloss_warned: diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index f60d09608..119dd2f63 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -17,13 +17,14 @@ from scrapy.utils.trackref import object_ref class Response(object_ref): - def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None): + def __init__(self, url, status=200, headers=None, body=b'', flags=None, request=None, certificate=None): self.headers = Headers(headers or {}) self.status = int(status) self._set_body(body) self._set_url(url) self.request = request self.flags = [] if flags is None else list(flags) + self.certificate = certificate @property def cb_kwargs(self): @@ -86,7 +87,7 @@ class Response(object_ref): """Create a new Response with the same attributes except for those given new values. """ - for x in ['url', 'status', 'headers', 'body', 'request', 'flags']: + for x in ['url', 'status', 'headers', 'body', 'request', 'flags', 'certificate']: kwargs.setdefault(x, getattr(self, x)) cls = kwargs.pop('cls', self.__class__) return cls(*args, **kwargs) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 64819acb6..bbe97d034 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -5,6 +5,7 @@ import sys from pytest import mark from testfixtures import LogCapture from twisted.internet import defer +from twisted.internet.ssl import Certificate from twisted.trial.unittest import TestCase from scrapy import signals @@ -407,3 +408,31 @@ with multiples lines yield crawler.crawl(self.mockserver.url("/status?n=200"), mockserver=self.mockserver) for req_id in range(3): self.assertIn("Got response 200, req_id %d" % req_id, str(log)) + + @defer.inlineCallbacks + def test_response_ssl_certificate_none(self): + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url("/echo?body=test", is_secure=False) + yield crawler.crawl(seed=url, mockserver=self.mockserver) + self.assertIsNone(crawler.spider.meta['responses'][0].certificate) + + @defer.inlineCallbacks + def test_response_ssl_certificate(self): + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url("/echo?body=test", is_secure=True) + yield crawler.crawl(seed=url, mockserver=self.mockserver) + cert = crawler.spider.meta['responses'][0].certificate + self.assertIsInstance(cert, Certificate) + self.assertEqual(cert.getSubject().commonName, b"localhost") + self.assertEqual(cert.getIssuer().commonName, b"localhost") + + @mark.xfail(reason="Responses with no body return early and contain no certificate") + @defer.inlineCallbacks + def test_response_ssl_certificate_empty_response(self): + crawler = self.runner.create_crawler(SingleRequestSpider) + url = self.mockserver.url("/status?n=200", is_secure=True) + yield crawler.crawl(seed=url, mockserver=self.mockserver) + cert = crawler.spider.meta['responses'][0].certificate + self.assertIsInstance(cert, Certificate) + self.assertEqual(cert.getSubject().commonName, b"localhost") + self.assertEqual(cert.getIssuer().commonName, b"localhost") From f85bf77da3c8943f0791dcae893e8294c4d118d7 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sun, 23 Feb 2020 18:31:13 -0300 Subject: [PATCH 228/313] Restore unrelated change --- scrapy/resolver.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index f69894b1e..554a3a14d 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -29,7 +29,7 @@ class CachingThreadedResolver(ThreadedResolver): cache_size = 0 return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) - def install_on_reactor(self): + def install_on_reactor(self,): self.reactor.installResolver(self) def getHostByName(self, name, timeout=None): From 889b4718520220d1a81e702ff754ec210a7d3c79 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sun, 23 Feb 2020 18:40:43 -0300 Subject: [PATCH 229/313] Import changes --- scrapy/core/downloader/handlers/http11.py | 4 ++-- tests/test_crawl.py | 4 +++- 2 files changed, 5 insertions(+), 3 deletions(-) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index e72275021..190ae1d3b 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -1,11 +1,11 @@ """Download handlers for http and https schemes""" +import ipaddress import logging import re import warnings from contextlib import suppress from io import BytesIO -from ipaddress import ip_address from time import time from urllib.parse import urldefrag @@ -468,7 +468,7 @@ class _ResponseReader(protocol.Protocol): self._certificate = ssl.Certificate(self.transport._producer.getPeerCertificate()) if self._ip_address is None: - self._ip_address = ip_address(self.transport._producer.getPeer().host) + self._ip_address = ipaddress.ip_address(self.transport._producer.getPeer().host) def dataReceived(self, bodyBytes): # This maybe called several times after cancel was called with buffered data. diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 3a9b00ab3..3c110e7a6 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -441,13 +441,15 @@ with multiples lines self.assertEqual(cert.getIssuer().commonName, b"localhost") @defer.inlineCallbacks - def test_dns_server_ip_address(self): + def test_dns_server_ip_address_none(self): crawler = self.runner.create_crawler(SingleRequestSpider) url = self.mockserver.url('/status?n=200') yield crawler.crawl(seed=url, mockserver=self.mockserver) ip_address = crawler.spider.meta['responses'][0].ip_address self.assertIsNone(ip_address) + @defer.inlineCallbacks + def test_dns_server_ip_address(self): crawler = self.runner.create_crawler(SingleRequestSpider) url = self.mockserver.url('/echo?body=test') expected_netloc, _ = urlparse(url).netloc.split(':') From 31f35c9c002178de20a3e124be4aab98c0a0f892 Mon Sep 17 00:00:00 2001 From: elacuesta <elacuesta@users.noreply.github.com> Date: Mon, 24 Feb 2020 08:02:00 -0300 Subject: [PATCH 230/313] Remove unnecessary comma (#4369) --- scrapy/resolver.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/resolver.py b/scrapy/resolver.py index 554a3a14d..f69894b1e 100644 --- a/scrapy/resolver.py +++ b/scrapy/resolver.py @@ -29,7 +29,7 @@ class CachingThreadedResolver(ThreadedResolver): cache_size = 0 return cls(reactor, cache_size, crawler.settings.getfloat('DNS_TIMEOUT')) - def install_on_reactor(self,): + def install_on_reactor(self): self.reactor.installResolver(self) def getHostByName(self, name, timeout=None): From 7417a9871c489686b7d3ac1b85b964eca253c979 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Mon, 24 Feb 2020 13:28:15 +0100 Subject: [PATCH 231/313] =?UTF-8?q?Make=20BaseItemExporter=E2=80=99s=20don?= =?UTF-8?q?t=5Ffail=20parameter=20keyword-only?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scrapy/exporters.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 96416f075..2e20a7180 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -23,7 +23,7 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter', class BaseItemExporter(object): - def __init__(self, dont_fail=False, **kwargs): + def __init__(self, *, dont_fail=False, **kwargs): self._kwargs = kwargs self._configure(kwargs, dont_fail=dont_fail) From a34c366fa4f226cb107a19561ca64fac0d1dbdd5 Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Fri, 21 Feb 2020 08:15:51 +0000 Subject: [PATCH 232/313] DOC linkcheck run; https and 301 link updates. Closes #4359 --- docs/conf.py | 1 + docs/contributing.rst | 6 +++--- docs/faq.rst | 6 +++--- docs/intro/install.rst | 12 ++++++------ docs/intro/tutorial.rst | 4 ++-- docs/news.rst | 19 +++++++++---------- docs/topics/broad-crawls.rst | 2 +- docs/topics/downloader-middleware.rst | 12 ++++++------ docs/topics/dynamic-content.rst | 6 +++--- docs/topics/item-pipeline.rst | 4 ++-- docs/topics/items.rst | 4 ++-- docs/topics/leaks.rst | 10 +++++----- docs/topics/request-response.rst | 2 +- docs/topics/selectors.rst | 17 ++++++++--------- docs/topics/shell.rst | 6 +++--- docs/topics/spiders.rst | 6 +++--- 16 files changed, 58 insertions(+), 59 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index c3418cfb3..6e2399f66 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -281,6 +281,7 @@ coverage_ignore_pyobjects = [ intersphinx_mapping = { 'coverage': ('https://coverage.readthedocs.io/en/stable', None), + 'cssselect': ('https://cssselect.readthedocs.io/en/latest', None), 'pytest': ('https://docs.pytest.org/en/latest', None), 'python': ('https://docs.python.org/3', None), 'sphinx': ('https://www.sphinx-doc.org/en/master', None), diff --git a/docs/contributing.rst b/docs/contributing.rst index f40a6bba2..aed5ab92e 100644 --- a/docs/contributing.rst +++ b/docs/contributing.rst @@ -143,7 +143,7 @@ by running ``git fetch upstream pull/$PR_NUMBER/head:$BRANCH_NAME_TO_CREATE`` (replace 'upstream' with a remote name for scrapy repository, ``$PR_NUMBER`` with an ID of the pull request, and ``$BRANCH_NAME_TO_CREATE`` with a name of the branch you want to create locally). -See also: https://help.github.com/articles/checking-out-pull-requests-locally/#modifying-an-inactive-pull-request-locally. +See also: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/checking-out-pull-requests-locally#modifying-an-inactive-pull-request-locally. When writing GitHub pull requests, try to keep titles short but descriptive. E.g. For bug #411: "Scrapy hangs if an exception raises in start_requests" @@ -168,7 +168,7 @@ Scrapy: * Don't put your name in the code you contribute; git provides enough metadata to identify author of the code. - See https://help.github.com/articles/setting-your-username-in-git/ for + See https://help.github.com/en/github/using-git/setting-your-username-in-git for setup instructions. .. _documentation-policies: @@ -266,5 +266,5 @@ And their unit-tests are in:: .. _tests/: https://github.com/scrapy/scrapy/tree/master/tests .. _open issues: https://github.com/scrapy/scrapy/issues .. _PEP 257: https://www.python.org/dev/peps/pep-0257/ -.. _pull request: https://help.github.com/en/articles/creating-a-pull-request +.. _pull request: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/creating-a-pull-request .. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist diff --git a/docs/faq.rst b/docs/faq.rst index f72e4cf01..75a0f4864 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -22,8 +22,8 @@ In other words, comparing `BeautifulSoup`_ (or `lxml`_) to Scrapy is like comparing `jinja2`_ to `Django`_. .. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/ -.. _lxml: http://lxml.de/ -.. _jinja2: http://jinja.pocoo.org/ +.. _lxml: https://lxml.de/ +.. _jinja2: https://palletsprojects.com/p/jinja/ .. _Django: https://www.djangoproject.com/ Can I use Scrapy with BeautifulSoup? @@ -269,7 +269,7 @@ The ``__VIEWSTATE`` parameter is used in sites built with ASP.NET/VB.NET. For more info on how it works see `this page`_. Also, here's an `example spider`_ which scrapes one of these sites. -.. _this page: http://search.cpan.org/~ecarroll/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm +.. _this page: https://metacpan.org/pod/release/ECARROLL/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm .. _example spider: https://github.com/AmbientLighter/rpn-fas/blob/master/fas/spiders/rnp.py What's the best way to parse big XML/CSV data feeds? diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 49968437c..871281460 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -65,7 +65,7 @@ please refer to their respective installation instructions: * `lxml installation`_ * `cryptography installation`_ -.. _lxml installation: http://lxml.de/installation.html +.. _lxml installation: https://lxml.de/installation.html .. _cryptography installation: https://cryptography.io/en/latest/installation/ @@ -253,11 +253,11 @@ For details, see `Issue #2473 <https://github.com/scrapy/scrapy/issues/2473>`_. .. _Python: https://www.python.org/ .. _pip: https://pip.pypa.io/en/latest/installing/ .. _lxml: https://lxml.de/index.html -.. _parsel: https://pypi.python.org/pypi/parsel -.. _w3lib: https://pypi.python.org/pypi/w3lib -.. _twisted: https://twistedmatrix.com/ -.. _cryptography: https://cryptography.io/ -.. _pyOpenSSL: https://pypi.python.org/pypi/pyOpenSSL +.. _parsel: https://pypi.org/project/parsel/ +.. _w3lib: https://pypi.org/project/w3lib/ +.. _twisted: https://twistedmatrix.com/trac/ +.. _cryptography: https://cryptography.io/en/latest/ +.. _pyOpenSSL: https://pypi.org/project/pyOpenSSL/ .. _setuptools: https://pypi.python.org/pypi/setuptools .. _AUR Scrapy package: https://aur.archlinux.org/packages/scrapy/ .. _homebrew: https://brew.sh/ diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 798fe4a7a..1768badbb 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -306,7 +306,7 @@ with a selector (see :ref:`topics-developer-tools`). visually selected elements, which works in many browsers. .. _regular expressions: https://docs.python.org/3/library/re.html -.. _Selector Gadget: http://selectorgadget.com/ +.. _Selector Gadget: https://selectorgadget.com/ XPath: a brief intro @@ -337,7 +337,7 @@ recommend `this tutorial to learn XPath through examples <http://zvon.org/comp/r/tut-XPath_1.html>`_, and `this tutorial to learn "how to think in XPath" <http://plasmasturm.org/log/xpath101/>`_. -.. _XPath: https://www.w3.org/TR/xpath +.. _XPath: https://www.w3.org/TR/xpath/all/ .. _CSS: https://www.w3.org/TR/selectors Extracting quotes and authors diff --git a/docs/news.rst b/docs/news.rst index e4b985c77..338b53dc4 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -26,7 +26,7 @@ Backward-incompatible changes * Python 3.4 is no longer supported, and some of the minimum requirements of Scrapy have also changed: - * cssselect_ 0.9.1 + * :doc:`cssselect <cssselect:index>` 0.9.1 * cryptography_ 2.0 * lxml_ 3.5.0 * pyOpenSSL_ 16.2.0 @@ -1616,7 +1616,7 @@ Deprecations and Removals + ``scrapy.utils.datatypes.SiteNode`` - The previously bundled ``scrapy.xlib.pydispatch`` library was deprecated and - replaced by `pydispatcher <https://pypi.python.org/pypi/PyDispatcher>`_. + replaced by `pydispatcher <https://pypi.org/project/PyDispatcher/>`_. Relocations @@ -2450,7 +2450,7 @@ Other ~~~~~ - Dropped Python 2.6 support (:issue:`448`) -- Add `cssselect`_ python package as install dependency +- Add :doc:`cssselect <cssselect:index>` python package as install dependency - Drop libxml2 and multi selector's backend support, `lxml`_ is required from now on. - Minimum Twisted version increased to 10.0.0, dropped Twisted 8.0 support. - Running test suite now requires ``mock`` python library (:issue:`390`) @@ -3047,17 +3047,16 @@ Scrapy 0.7 First release of Scrapy. -.. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1 +.. _AJAX crawleable urls: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started?csw=1 .. _botocore: https://github.com/boto/botocore .. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding .. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/ .. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request .. _cryptography: https://cryptography.io/en/latest/ -.. _cssselect: https://github.com/scrapy/cssselect/ -.. _docstrings: https://docs.python.org/glossary.html#term-docstring -.. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt +.. _docstrings: https://docs.python.org/3/glossary.html#term-docstring +.. _KeyboardInterrupt: https://docs.python.org/3/library/exceptions.html#KeyboardInterrupt .. _LevelDB: https://github.com/google/leveldb -.. _lxml: http://lxml.de/ +.. _lxml: https://lxml.de/ .. _marshal: https://docs.python.org/2/library/marshal.html .. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator .. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator @@ -3068,11 +3067,11 @@ First release of Scrapy. .. _queuelib: https://github.com/scrapy/queuelib .. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml .. _resource: https://docs.python.org/2/library/resource.html -.. _robots.txt: http://www.robotstxt.org/ +.. _robots.txt: https://www.robotstxt.org/ .. _scrapely: https://github.com/scrapy/scrapely .. _service_identity: https://service-identity.readthedocs.io/en/stable/ .. _six: https://six.readthedocs.io/ -.. _tox: https://pypi.python.org/pypi/tox +.. _tox: https://pypi.org/project/tox/ .. _Twisted: https://twistedmatrix.com/trac/ .. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/ .. _w3lib: https://github.com/scrapy/w3lib diff --git a/docs/topics/broad-crawls.rst b/docs/topics/broad-crawls.rst index 4922694ee..63b60312e 100644 --- a/docs/topics/broad-crawls.rst +++ b/docs/topics/broad-crawls.rst @@ -188,7 +188,7 @@ AjaxCrawlMiddleware helps to crawl them correctly. It is turned OFF by default because it has some performance overhead, and enabling it for focused crawls doesn't make much sense. -.. _ajax crawlable: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started +.. _ajax crawlable: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started .. _broad-crawls-bfo: diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index a83cedcfd..0297ef3a0 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -709,7 +709,7 @@ HttpCompressionMiddleware provided `brotlipy`_ is installed. .. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt -.. _brotlipy: https://pypi.python.org/pypi/brotlipy +.. _brotlipy: https://pypi.org/project/brotlipy/ HttpCompressionMiddleware Settings ~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ @@ -1038,7 +1038,7 @@ Based on `RobotFileParser * is Python's built-in robots.txt_ parser * is compliant with `Martijn Koster's 1996 draft specification - <http://www.robotstxt.org/norobots-rfc.txt>`_ + <https://www.robotstxt.org/norobots-rfc.txt>`_ * lacks support for wildcard matching @@ -1061,7 +1061,7 @@ Based on `Reppy <https://github.com/seomoz/reppy/>`_: <https://github.com/seomoz/rep-cpp>`_ * is compliant with `Martijn Koster's 1996 draft specification - <http://www.robotstxt.org/norobots-rfc.txt>`_ + <https://www.robotstxt.org/norobots-rfc.txt>`_ * supports wildcard matching @@ -1086,7 +1086,7 @@ Based on `Robotexclusionrulesparser <http://nikitathespider.com/python/rerp/>`_: * implemented in Python * is compliant with `Martijn Koster's 1996 draft specification - <http://www.robotstxt.org/norobots-rfc.txt>`_ + <https://www.robotstxt.org/norobots-rfc.txt>`_ * supports wildcard matching @@ -1115,7 +1115,7 @@ implementing the methods described below. .. autoclass:: RobotParser :members: -.. _robots.txt: http://www.robotstxt.org/ +.. _robots.txt: https://www.robotstxt.org/ DownloaderStats --------------- @@ -1155,7 +1155,7 @@ AjaxCrawlMiddleware Middleware that finds 'AJAX crawlable' page variants based on meta-fragment html tag. See - https://developers.google.com/webmasters/ajax-crawling/docs/getting-started + https://developers.google.com/search/docs/ajax-crawling/docs/getting-started for more info. .. note:: diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index 1c3607860..b98133676 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -241,12 +241,12 @@ along with `scrapy-selenium`_ for seamless integration. .. _headless browser: https://en.wikipedia.org/wiki/Headless_browser .. _JavaScript: https://en.wikipedia.org/wiki/JavaScript .. _js2xml: https://github.com/scrapinghub/js2xml -.. _json.loads: https://docs.python.org/library/json.html#json.loads +.. _json.loads: https://docs.python.org/3/library/json.html#json.loads .. _pytesseract: https://github.com/madmaze/pytesseract -.. _regular expression: https://docs.python.org/library/re.html +.. _regular expression: https://docs.python.org/3/library/re.html .. _scrapy-selenium: https://github.com/clemfromspace/scrapy-selenium .. _scrapy-splash: https://github.com/scrapy-plugins/scrapy-splash -.. _Selenium: https://www.seleniumhq.org/ +.. _Selenium: https://www.selenium.dev/ .. _Splash: https://github.com/scrapinghub/splash .. _tabula-py: https://github.com/chezou/tabula-py .. _wget: https://www.gnu.org/software/wget/ diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index cdc4953c2..801d48fd5 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -158,8 +158,8 @@ method and how to clean up the resources properly.:: self.db[self.collection_name].insert_one(dict(item)) return item -.. _MongoDB: https://www.mongodb.org/ -.. _pymongo: https://api.mongodb.org/python/current/ +.. _MongoDB: https://www.mongodb.com/ +.. _pymongo: https://api.mongodb.com/python/current/ Take screenshot of item diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 15313775b..44643cb67 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -166,7 +166,7 @@ If your item contains mutable_ values like lists or dictionaries, a shallow copy will keep references to the same mutable values across all different copies. -.. _mutable: https://docs.python.org/glossary.html#term-mutable +.. _mutable: https://docs.python.org/3/glossary.html#term-mutable For example, if you have an item with a list of tags, and you create a shallow copy of that item, both the original item and the copy have the same list of @@ -177,7 +177,7 @@ If that is not the desired behavior, use a deep copy instead. See the `documentation of the copy module`_ for more information. -.. _documentation of the copy module: https://docs.python.org/library/copy.html +.. _documentation of the copy module: https://docs.python.org/3/library/copy.html To create a shallow copy of an item, you can either call :meth:`~scrapy.item.Item.copy` on an existing item diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 9fee333ac..c0c83fc84 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -206,7 +206,7 @@ objects. If this is your case, and you can't find your leaks using ``trackref``, you still have another resource: the `Guppy library`_. If you're using Python3, see :ref:`topics-leaks-muppy`. -.. _Guppy library: https://pypi.python.org/pypi/guppy +.. _Guppy library: https://pypi.org/project/guppy/ If you use ``pip``, you can install Guppy with the following command:: @@ -311,9 +311,9 @@ though neither Scrapy nor your project are leaking memory. This is due to a (not so well) known problem of Python, which may not return released memory to the operating system in some cases. For more information on this issue see: -* `Python Memory Management <http://www.evanjones.ca/python-memory.html>`_ -* `Python Memory Management Part 2 <http://www.evanjones.ca/python-memory-part2.html>`_ -* `Python Memory Management Part 3 <http://www.evanjones.ca/python-memory-part3.html>`_ +* `Python Memory Management <https://www.evanjones.ca/python-memory.html>`_ +* `Python Memory Management Part 2 <https://www.evanjones.ca/python-memory-part2.html>`_ +* `Python Memory Management Part 3 <https://www.evanjones.ca/python-memory-part3.html>`_ The improvements proposed by Evan Jones, which are detailed in `this paper`_, got merged in Python 2.5, but this only reduces the problem, it doesn't fix it @@ -327,7 +327,7 @@ completely. To quote the paper: to move to a compacting garbage collector, which is able to move objects in memory. This would require significant changes to the Python interpreter.* -.. _this paper: http://www.evanjones.ca/memoryallocator/ +.. _this paper: https://www.evanjones.ca/memoryallocator/ To keep memory consumption reasonable you can split the job into several smaller jobs or enable :ref:`persistent job queue <topics-jobs>` diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index f009facd6..c4c2845c9 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -396,7 +396,7 @@ The FormRequest class extends the base :class:`Request` with functionality for dealing with HTML forms. It uses `lxml.html forms`_ to pre-populate form fields with form data from :class:`Response` objects. -.. _lxml.html forms: http://lxml.de/lxmlhtml.html#forms +.. _lxml.html forms: https://lxml.de/lxmlhtml.html#forms .. class:: FormRequest(url, [formdata, ...]) diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index c3d431e2a..1f7802c98 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -35,12 +35,11 @@ defines selectors to associate those styles with specific HTML elements. in speed and parsing accuracy to lxml. .. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/ -.. _lxml: http://lxml.de/ +.. _lxml: https://lxml.de/ .. _ElementTree: https://docs.python.org/2/library/xml.etree.elementtree.html -.. _cssselect: https://pypi.python.org/pypi/cssselect/ -.. _XPath: https://www.w3.org/TR/xpath +.. _XPath: https://www.w3.org/TR/xpath/all/ .. _CSS: https://www.w3.org/TR/selectors -.. _parsel: https://parsel.readthedocs.io/ +.. _parsel: https://parsel.readthedocs.io/en/latest/ Using selectors =============== @@ -255,7 +254,7 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**: They will most probably not work with other libraries like `lxml`_ or `PyQuery`_. -.. _PyQuery: https://pypi.python.org/pypi/pyquery +.. _PyQuery: https://pypi.org/project/pyquery/ Examples: @@ -309,7 +308,7 @@ Examples: make much sense: text nodes do not have attributes, and attribute values are string values already and do not have children nodes. -.. _CSS Selectors: https://www.w3.org/TR/css3-selectors/#selectors +.. _CSS Selectors: https://www.w3.org/TR/selectors-3/#selectors .. _topics-selectors-nesting-selectors: @@ -504,7 +503,7 @@ Another common case would be to extract all direct ``<p>`` children: For more details about relative XPaths see the `Location Paths`_ section in the XPath specification. -.. _Location Paths: https://www.w3.org/TR/xpath#location-paths +.. _Location Paths: https://www.w3.org/TR/xpath/all/#location-paths When querying by class, consider using CSS ------------------------------------------ @@ -612,7 +611,7 @@ But using the ``.`` to mean the node, works: >>> sel.xpath("//a[contains(., 'Next Page')]").getall() ['<a href="#">Click here to go to the <strong>Next Page</strong></a>'] -.. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions +.. _`XPath string function`: https://www.w3.org/TR/xpath/all/#section-String-Functions .. _topics-selectors-xpath-variables: @@ -764,7 +763,7 @@ Set operations These can be handy for excluding parts of a document tree before extracting text elements for example. -Example extracting microdata (sample content taken from http://schema.org/Product) +Example extracting microdata (sample content taken from https://schema.org/Product) with groups of itemscopes and corresponding itemprops:: >>> doc = u""" diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index 3cf8311a6..8f7518b19 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -41,7 +41,7 @@ variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`:: .. _IPython: https://ipython.org/ .. _IPython installation guide: https://ipython.org/install.html -.. _bpython: https://www.bpython-interpreter.org/ +.. _bpython: https://bpython-interpreter.org/ Launch the shell ================ @@ -142,7 +142,7 @@ Example of shell session ======================== Here's an example of a typical shell session where we start by scraping the -https://scrapy.org page, and then proceed to scrape the https://reddit.com +https://scrapy.org page, and then proceed to scrape the https://old.reddit.com/ page. Finally, we modify the (Reddit) request method to POST and re-fetch it getting an error. We end the session by typing Ctrl-D (in Unix systems) or Ctrl-Z in Windows. @@ -182,7 +182,7 @@ After that, we can start playing with the objects: >>> response.xpath('//title/text()').get() 'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' ->>> fetch("https://reddit.com") +>>> fetch("https://old.reddit.com/") >>> response.xpath('//title/text()').get() 'reddit: the front page of the internet' diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index b0fb14e24..e0f33de66 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -299,8 +299,8 @@ The spider will not do any parsing on its own. If you were to set the ``start_urls`` attribute from the command line, you would have to parse it on your own into a list using something like -`ast.literal_eval <https://docs.python.org/library/ast.html#ast.literal_eval>`_ -or `json.loads <https://docs.python.org/library/json.html#json.loads>`_ +`ast.literal_eval <https://docs.python.org/3/library/ast.html#ast.literal_eval>`_ +or `json.loads <https://docs.python.org/3/library/json.html#json.loads>`_ and then set it as an attribute. Otherwise, you would cause iteration over a ``start_urls`` string (a very common python pitfall) @@ -811,6 +811,6 @@ Combine SitemapSpider with other sources of urls:: .. _Sitemaps: https://www.sitemaps.org/index.html .. _Sitemap index files: https://www.sitemaps.org/protocol.html#index -.. _robots.txt: http://www.robotstxt.org/ +.. _robots.txt: https://www.robotstxt.org/ .. _TLD: https://en.wikipedia.org/wiki/Top-level_domain .. _Scrapyd documentation: https://scrapyd.readthedocs.io/en/latest/ From 034e2c31c7d55333c3de208f80dcee1bf45ef9b9 Mon Sep 17 00:00:00 2001 From: gunblues <hsiao.powen@gmail.com> Date: Wed, 26 Feb 2020 03:46:05 +0800 Subject: [PATCH 233/313] Use a non-zero exit code when a pipeline's open_spider method throws an exception (#4207) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * fix issue 4175 - Scrapy does not use a non-zero exit code when pipeline's open_spider throws the exception * remove extra blank lines * remove redundant code * remove blank line at end of file * more suitable naming for response and make if-condition shorter * avoid error - AttributeError: 'Deferred' object has no attribute 'result' * use getattr to make code concisely * add test * remove useless file * modify test class name * remove unneccessary files * Fix Flake8-reported issue * fix these items which are suggested by Gallaecio ・Sort those imports at tests/test_cmdline_crawl_with_pipeline/__init__.py ・Remove the unused setUp method. ・Remove comments generated by Scrapy’s project generation tool. ・Remove the [deploy] section from the scrapy.cfg file (I don’t think it’s needed here) ・Remove BOT_NAME and NEWSPIDER_MODULE from settings.py (I think there are not needed either, although I’m less sure about NEWSPIDER_MODULE) * have to reserve BOT_NAME, SPIDER_MODULES in settings.py * Remove unneeded empty lines * Empty __init__.py file with unneeded comments * Remove an unneeded empty line at the end * Remove unneeed empty line from __init__.py file * Update __init__.py * Update __init__.py * Update exception.py * Update normal.py * Update __init__.py * Update __init__.py * fix W391 blank line at end of file Co-authored-by: Adrián Chaves <adrian@chaves.io> --- scrapy/commands/crawl.py | 11 +++++++--- .../__init__.py | 20 +++++++++++++++++++ .../scrapy.cfg | 2 ++ .../test_spider/__init__.py | 0 .../test_spider/pipelines.py | 16 +++++++++++++++ .../test_spider/settings.py | 2 ++ .../test_spider/spiders/__init__.py | 0 .../test_spider/spiders/exception.py | 14 +++++++++++++ .../test_spider/spiders/normal.py | 14 +++++++++++++ 9 files changed, 76 insertions(+), 3 deletions(-) create mode 100644 tests/test_cmdline_crawl_with_pipeline/__init__.py create mode 100644 tests/test_cmdline_crawl_with_pipeline/scrapy.cfg create mode 100644 tests/test_cmdline_crawl_with_pipeline/test_spider/__init__.py create mode 100644 tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py create mode 100644 tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py create mode 100644 tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/__init__.py create mode 100644 tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py create mode 100644 tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index 8093fd402..7b417e2eb 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -54,8 +54,13 @@ class Command(ScrapyCommand): raise UsageError("running 'scrapy crawl' with more than one spider is no longer supported") spname = args[0] - self.crawler_process.crawl(spname, **opts.spargs) - self.crawler_process.start() + crawl_defer = self.crawler_process.crawl(spname, **opts.spargs) - if self.crawler_process.bootstrap_failed: + if getattr(crawl_defer, 'result', None) is not None and issubclass(crawl_defer.result.type, Exception): self.exitcode = 1 + else: + self.crawler_process.start() + + if self.crawler_process.bootstrap_failed or \ + (hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception): + self.exitcode = 1 diff --git a/tests/test_cmdline_crawl_with_pipeline/__init__.py b/tests/test_cmdline_crawl_with_pipeline/__init__.py new file mode 100644 index 000000000..d341888d3 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/__init__.py @@ -0,0 +1,20 @@ +import os +import sys +import unittest +from subprocess import Popen, PIPE + + +class CmdlineCrawlPipelineTest(unittest.TestCase): + + def _execute(self, spname): + args = (sys.executable, '-m', 'scrapy.cmdline', 'crawl', spname) + cwd = os.path.dirname(os.path.abspath(__file__)) + proc = Popen(args, stdout=PIPE, stderr=PIPE, cwd=cwd) + proc.communicate() + return proc.returncode + + def test_open_spider_normally_in_pipeline(self): + self.assertEqual(self._execute('normal'), 0) + + def test_exception_at_open_spider_in_pipeline(self): + self.assertEqual(self._execute('exception'), 1) diff --git a/tests/test_cmdline_crawl_with_pipeline/scrapy.cfg b/tests/test_cmdline_crawl_with_pipeline/scrapy.cfg new file mode 100644 index 000000000..2f238dba3 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/scrapy.cfg @@ -0,0 +1,2 @@ +[settings] +default = test_spider.settings diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/__init__.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py new file mode 100644 index 000000000..ce916f699 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py @@ -0,0 +1,16 @@ +class TestSpiderPipeline(object): + + def open_spider(self, spider): + pass + + def process_item(self, item, spider): + return item + + +class TestSpiderExceptionPipeline(object): + + def open_spider(self, spider): + raise Exception('exception') + + def process_item(self, item, spider): + return item diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py new file mode 100644 index 000000000..ae782c0d8 --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/settings.py @@ -0,0 +1,2 @@ +BOT_NAME = 'test_spider' +SPIDER_MODULES = ['test_spider.spiders'] diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/__init__.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/__init__.py new file mode 100644 index 000000000..e69de29bb diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py new file mode 100644 index 000000000..300f45ebf --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/exception.py @@ -0,0 +1,14 @@ +import scrapy + + +class ExceptionSpider(scrapy.Spider): + name = 'exception' + + custom_settings = { + 'ITEM_PIPELINES': { + 'test_spider.pipelines.TestSpiderExceptionPipeline': 300 + } + } + + def parse(self, response): + pass diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py new file mode 100644 index 000000000..87a40fdcb --- /dev/null +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/spiders/normal.py @@ -0,0 +1,14 @@ +import scrapy + + +class NormalSpider(scrapy.Spider): + name = 'normal' + + custom_settings = { + 'ITEM_PIPELINES': { + 'test_spider.pipelines.TestSpiderPipeline': 300 + } + } + + def parse(self, response): + pass From 7291173f6b6a8e1768ab9d5f52474cd8ada8381e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Tue, 25 Feb 2020 21:35:21 +0100 Subject: [PATCH 234/313] Have ReadTheDocs builds fail on warning --- .readthedocs.yml | 1 + 1 file changed, 1 insertion(+) diff --git a/.readthedocs.yml b/.readthedocs.yml index 563add75f..0b9e15018 100644 --- a/.readthedocs.yml +++ b/.readthedocs.yml @@ -1,6 +1,7 @@ version: 2 sphinx: configuration: docs/conf.py + fail_on_warning: true python: # For available versions, see: # https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image From a9d7d8f064fe3086227d4737d07e3ca4a296b4b2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Tue, 25 Feb 2020 21:41:07 +0100 Subject: [PATCH 235/313] Add Scrapy dependencies back to docs/requirements.txt --- docs/requirements.txt | 15 +++++++++++++++ setup.py | 1 + 2 files changed, 16 insertions(+) diff --git a/docs/requirements.txt b/docs/requirements.txt index 773b92cea..215cdd64d 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -2,3 +2,18 @@ Sphinx>=2.1 sphinx-hoverxref sphinx-notfound-page sphinx_rtd_theme + +# Required for ReadTheDocs +# Keep in sync with setup.py +Twisted>=17.9.0 +cryptography>=2.0 +cssselect>=0.9.1 +lxml>=3.5.0 +parsel>=1.5.0 +PyDispatcher>=2.0.5 +pyOpenSSL>=16.2.0 +queuelib>=1.4.2 +service_identity>=16.0.0 +w3lib>=1.17.0 +zope.interface>=4.1.3 +protego>=0.1.15 diff --git a/setup.py b/setup.py index 85d797f88..6f15ca277 100644 --- a/setup.py +++ b/setup.py @@ -62,6 +62,7 @@ setup( 'Topic :: Software Development :: Libraries :: Python Modules', ], python_requires='>=3.5', + # Keep in sync with docs/requirements.txt install_requires=[ 'Twisted>=17.9.0', 'cryptography>=2.0', From 778813717df0d5fcd4359266f6785068feea0785 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Tue, 25 Feb 2020 21:58:28 +0100 Subject: [PATCH 236/313] Use ReadTheDocs install.path --- .readthedocs.yml | 1 + docs/requirements.txt | 15 --------------- setup.py | 1 - 3 files changed, 1 insertion(+), 16 deletions(-) diff --git a/.readthedocs.yml b/.readthedocs.yml index 0b9e15018..17eba34f3 100644 --- a/.readthedocs.yml +++ b/.readthedocs.yml @@ -8,3 +8,4 @@ python: version: 3.7 # Keep in sync with .travis.yml install: - requirements: docs/requirements.txt + - path: . diff --git a/docs/requirements.txt b/docs/requirements.txt index 215cdd64d..773b92cea 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -2,18 +2,3 @@ Sphinx>=2.1 sphinx-hoverxref sphinx-notfound-page sphinx_rtd_theme - -# Required for ReadTheDocs -# Keep in sync with setup.py -Twisted>=17.9.0 -cryptography>=2.0 -cssselect>=0.9.1 -lxml>=3.5.0 -parsel>=1.5.0 -PyDispatcher>=2.0.5 -pyOpenSSL>=16.2.0 -queuelib>=1.4.2 -service_identity>=16.0.0 -w3lib>=1.17.0 -zope.interface>=4.1.3 -protego>=0.1.15 diff --git a/setup.py b/setup.py index 6f15ca277..85d797f88 100644 --- a/setup.py +++ b/setup.py @@ -62,7 +62,6 @@ setup( 'Topic :: Software Development :: Libraries :: Python Modules', ], python_requires='>=3.5', - # Keep in sync with docs/requirements.txt install_requires=[ 'Twisted>=17.9.0', 'cryptography>=2.0', From 6109ad9aacd5897f67842685b2405627b4af4ad6 Mon Sep 17 00:00:00 2001 From: HEndo12345 <38522238+HEndo12345@users.noreply.github.com> Date: Thu, 27 Feb 2020 23:15:30 +0900 Subject: [PATCH 237/313] Clean up the deprecated settings list (#4378) --- scrapy/settings/deprecated.py | 2 -- 1 file changed, 2 deletions(-) diff --git a/scrapy/settings/deprecated.py b/scrapy/settings/deprecated.py index 91ed689e8..1211908df 100644 --- a/scrapy/settings/deprecated.py +++ b/scrapy/settings/deprecated.py @@ -9,10 +9,8 @@ DEPRECATED_SETTINGS = [ ('ENCODING_ALIASES', 'no longer needed (encoding discovery uses w3lib now)'), ('STATS_ENABLED', 'no longer supported (change STATS_CLASS instead)'), ('SQLITE_DB', 'no longer supported'), - ('SELECTORS_BACKEND', 'use SCRAPY_SELECTORS_BACKEND environment variable instead'), ('AUTOTHROTTLE_MIN_DOWNLOAD_DELAY', 'use DOWNLOAD_DELAY instead'), ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), - ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), ('REDIRECT_MAX_METAREFRESH_DELAY', 'use METAREFRESH_MAXDELAY instead'), ('LOG_UNSERIALIZABLE_REQUESTS', 'use SCHEDULER_DEBUG instead'), ] From 2acaa86231e8a743333928907c2933feadf40cd7 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 27 Feb 2020 15:39:49 +0100 Subject: [PATCH 238/313] Do not warn about valid environment variables --- scrapy/utils/project.py | 25 ++++++++++++++++++++----- 1 file changed, 20 insertions(+), 5 deletions(-) diff --git a/scrapy/utils/project.py b/scrapy/utils/project.py index d9a03ff63..d1dec2543 100644 --- a/scrapy/utils/project.py +++ b/scrapy/utils/project.py @@ -75,9 +75,24 @@ def get_project_settings(): "is deprecated.", ScrapyDeprecationWarning) settings.setdict(pickle.loads(pickled_settings), priority='project') - env_overrides = {k[7:]: v for k, v in os.environ.items() if - k.startswith('SCRAPY_')} - if env_overrides: - warnings.warn("Use of 'SCRAPY_'-prefixed environment variables to override settings is deprecated.", ScrapyDeprecationWarning) - settings.setdict(env_overrides, priority='project') + scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if + k.startswith('SCRAPY_')} + valid_envvars = { + 'SCRAPY_CHECK', + 'SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE', + 'SCRAPY_PROJECT', + 'SCRAPY_PYTHON_SHELL', + 'SCRAPY_SETTINGS_MODULE', + } + setting_envvars = {k for k in scrapy_envvars if k not in valid_envvars} + if setting_envvars: + setting_envvar_list = ', '.join(sorted(setting_envvars)) + warnings.warn( + 'Use of environment variables prefixed with SCRAPY_ to override ' + 'settings is deprecated. The following environment variables are ' + 'currently defined: {}'.format(setting_envvar_list), + ScrapyDeprecationWarning + ) + settings.setdict(scrapy_envvars, priority='project') + return settings From 9aae4c0be7b42e27daa2750b6f01eb497edcd98a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Thu, 27 Feb 2020 16:31:43 +0100 Subject: [PATCH 239/313] Add tests for envvar setting warnings --- scrapy/utils/project.py | 10 +++---- tests/test_utils_project.py | 56 ++++++++++++++++++++++++++++++++++++- 2 files changed, 60 insertions(+), 6 deletions(-) diff --git a/scrapy/utils/project.py b/scrapy/utils/project.py index d1dec2543..b8d3ebf9d 100644 --- a/scrapy/utils/project.py +++ b/scrapy/utils/project.py @@ -78,11 +78,11 @@ def get_project_settings(): scrapy_envvars = {k[7:]: v for k, v in os.environ.items() if k.startswith('SCRAPY_')} valid_envvars = { - 'SCRAPY_CHECK', - 'SCRAPY_PICKLED_SETTINGS_TO_OVERRIDE', - 'SCRAPY_PROJECT', - 'SCRAPY_PYTHON_SHELL', - 'SCRAPY_SETTINGS_MODULE', + 'CHECK', + 'PICKLED_SETTINGS_TO_OVERRIDE', + 'PROJECT', + 'PYTHON_SHELL', + 'SETTINGS_MODULE', } setting_envvars = {k for k in scrapy_envvars if k not in valid_envvars} if setting_envvars: diff --git a/tests/test_utils_project.py b/tests/test_utils_project.py index bd74b0c34..1ef4eeb14 100644 --- a/tests/test_utils_project.py +++ b/tests/test_utils_project.py @@ -3,7 +3,11 @@ import os import tempfile import shutil import contextlib -from scrapy.utils.project import data_path + +from pytest import warns + +from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.utils.project import data_path, get_project_settings @contextlib.contextmanager @@ -41,3 +45,53 @@ class ProjectUtilsTest(unittest.TestCase): ) abspath = os.path.join(os.path.sep, 'absolute', 'path') self.assertEqual(abspath, data_path(abspath)) + + +@contextlib.contextmanager +def set_env(**update): + modified = set(update.keys()) & set(os.environ.keys()) + update_after = {k: os.environ[k] for k in modified} + remove_after = frozenset(k for k in update if k not in os.environ) + try: + os.environ.update(update) + yield + finally: + os.environ.update(update_after) + for k in remove_after: + os.environ.pop(k) + + +class GetProjectSettingsTestCase(unittest.TestCase): + + def test_valid_envvar(self): + value = 'tests.test_cmdline.settings' + envvars = { + 'SCRAPY_SETTINGS_MODULE': value, + } + with set_env(**envvars), warns(None) as warnings: + settings = get_project_settings() + assert not warnings + assert settings.get('SETTINGS_MODULE') == value + + def test_invalid_envvar(self): + envvars = { + 'SCRAPY_FOO': 'bar', + } + with set_env(**envvars), warns(None) as warnings: + get_project_settings() + assert len(warnings) == 1 + assert warnings[0].category == ScrapyDeprecationWarning + assert str(warnings[0].message).endswith(': FOO') + + def test_valid_and_invalid_envvars(self): + value = 'tests.test_cmdline.settings' + envvars = { + 'SCRAPY_FOO': 'bar', + 'SCRAPY_SETTINGS_MODULE': value, + } + with set_env(**envvars), warns(None) as warnings: + settings = get_project_settings() + assert len(warnings) == 1 + assert warnings[0].category == ScrapyDeprecationWarning + assert str(warnings[0].message).endswith(': FOO') + assert settings.get('SETTINGS_MODULE') == value From c411a51f42a5e6d241d69349b228f1584fdbd31b Mon Sep 17 00:00:00 2001 From: sakshamb2113 <44064539+sakshamb2113@users.noreply.github.com> Date: Fri, 28 Feb 2020 17:47:02 +0530 Subject: [PATCH 240/313] Fix random failures from test_fixed_delay in some machines (#4372) --- tests/test_crawl.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index bbe97d034..e93c668c5 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -37,7 +37,7 @@ class CrawlTestCase(TestCase): @defer.inlineCallbacks def test_fixed_delay(self): - yield self._test_delay(total=3, delay=0.1) + yield self._test_delay(total=3, delay=0.2) @defer.inlineCallbacks def test_randomized_delay(self): From ef00f8eb8eb4f5727409fd40c5826661db2bb665 Mon Sep 17 00:00:00 2001 From: MaliCN <40772522+MaliYudina@users.noreply.github.com> Date: Fri, 28 Feb 2020 22:42:07 +0300 Subject: [PATCH 241/313] updated with new macOS name (#4308) (#4323) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * changed for new name as "macOS" (issue #4308) * updated macOS name * update macOS name * updated macOS name * update for new macOS name * docs/intro/install.rst: fix macOS header symbols Co-Authored-By: elacuesta <elacuesta@users.noreply.github.com> Co-authored-by: Adrián Chaves <adrian@chaves.io> Co-authored-by: elacuesta <elacuesta@users.noreply.github.com> --- README.rst | 2 +- docs/intro/install.rst | 12 ++++++------ docs/news.rst | 12 ++++++------ scrapy/extensions/memusage.py | 2 +- 4 files changed, 14 insertions(+), 14 deletions(-) diff --git a/README.rst b/README.rst index 7fefaeec9..ce5973bcd 100644 --- a/README.rst +++ b/README.rst @@ -41,7 +41,7 @@ Requirements ============ * Python 3.5+ -* Works on Linux, Windows, Mac OSX, BSD +* Works on Linux, Windows, macOS, BSD Install ======= diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 871281460..89ba0c154 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -12,7 +12,7 @@ under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). If you're using `Anaconda`_ or `Miniconda`_, you can install the package from the `conda-forge`_ channel, which has up-to-date packages for Linux, Windows -and OS X. +and macOS. To install Scrapy using ``conda``, run:: @@ -148,11 +148,11 @@ you can install Scrapy with ``pip`` after that:: .. _intro-install-macos: -Mac OS X --------- +macOS +----- Building Scrapy's dependencies requires the presence of a C compiler and -development headers. On OS X this is typically provided by Apple’s Xcode +development headers. On macOS this is typically provided by Apple’s Xcode development tools. To install the Xcode command line tools open a terminal window and run:: @@ -191,7 +191,7 @@ solutions: * *(Optional)* :ref:`Install Scrapy inside a Python virtual environment <intro-using-virtualenv>`. - This method is a workaround for the above OS X issue, but it's an overall + This method is a workaround for the above macOS issue, but it's an overall good practice for managing dependencies and can complement the first method. After any of these workarounds you should be able to install Scrapy:: @@ -207,7 +207,7 @@ For PyPy3, only Linux installation was tested. Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy. This means that these dependecies will be built during installation. -On OS X, you are likely to face an issue with building Cryptography dependency, +On macOS, you are likely to face an issue with building Cryptography dependency, solution to this problem is described `here <https://github.com/pyca/cryptography/issues/2692#issuecomment-272773481>`_, that is to ``brew install openssl`` and then export the flags that this command diff --git a/docs/news.rst b/docs/news.rst index 338b53dc4..c1daedaf2 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -1076,7 +1076,7 @@ Cleanups & Refactoring ~~~~~~~~~~~~~~~~~~~~~~ - Tests: remove temp files and folders (:issue:`2570`), - fixed ProjectUtilsTest on OS X (:issue:`2569`), + fixed ProjectUtilsTest on macOS (:issue:`2569`), use portable pypy for Linux on Travis CI (:issue:`2710`) - Separate building request from ``_requests_to_follow`` in CrawlSpider (:issue:`2562`) - Remove “Python 3 progress” badge (:issue:`2567`) @@ -1645,7 +1645,7 @@ Bugfixes - Makes ``_monkeypatches`` more robust (:issue:`1634`). - Fixed bug on ``XMLItemExporter`` with non-string fields in items (:issue:`1738`). -- Fixed startproject command in OS X (:issue:`1635`). +- Fixed startproject command in macOS (:issue:`1635`). - Fixed :class:`~scrapy.exporters.PythonItemExporter` and CSVExporter for non-string item types (:issue:`1737`). - Various logging related fixes (:issue:`1294`, :issue:`1419`, :issue:`1263`, @@ -1713,12 +1713,12 @@ Scrapy 1.0.4 (2015-12-30) - Typos corrections (:commit:`7067117`) - fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`) - Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`) -- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`) +- Replace alternative macOS install workaround with virtualenv (:commit:`98b63ee`) - Reference Homebrew's homepage for installation instructions (:commit:`1925db1`) - Add oldest supported tox version to contributing docs (:commit:`5d10d6d`) - Note in install docs about pip being already included in python>=2.7.9 (:commit:`85c980e`) - Add non-python dependencies to Ubuntu install section in the docs (:commit:`fbd010d`) -- Add OS X installation section to docs (:commit:`d8f4cba`) +- Add macOS installation section to docs (:commit:`d8f4cba`) - DOC(ENH): specify path to rtd theme explicitly (:commit:`de73b1a`) - minor: scrapy.Spider docs grammar (:commit:`1ddcc7b`) - Make common practices sample code match the comments (:commit:`1b85bcf`) @@ -2571,7 +2571,7 @@ Scrapy 0.18.0 (released 2013-08-09) - MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62 - added from_crawler method to spiders - added system tests with mock server -- more improvements to Mac OS compatibility (thanks Alex Cepoi) +- more improvements to macOS compatibility (thanks Alex Cepoi) - several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez) - support custom download slots - added --spider option to "shell" command. @@ -2647,7 +2647,7 @@ Scrapy 0.16.3 (released 2012-12-07) - Remove concurrency limitation when using download delays and still ensure inter-request delays are enforced (:commit:`487b9b5`) - add error details when image pipeline fails (:commit:`8232569`) -- improve mac os compatibility (:commit:`8dcf8aa`) +- improve macOS compatibility (:commit:`8dcf8aa`) - setup.py: use README.rst to populate long_description (:commit:`7b5310d`) - doc: removed obsolete references to ClientForm (:commit:`80f9bb6`) - correct docs for default storage backend (:commit:`2aa491b`) diff --git a/scrapy/extensions/memusage.py b/scrapy/extensions/memusage.py index c0570567e..14e0fb32d 100644 --- a/scrapy/extensions/memusage.py +++ b/scrapy/extensions/memusage.py @@ -47,7 +47,7 @@ class MemoryUsage(object): def get_virtual_size(self): size = self.resource.getrusage(self.resource.RUSAGE_SELF).ru_maxrss if sys.platform != 'darwin': - # on Mac OS X ru_maxrss is in bytes, on Linux it is in KB + # on macOS ru_maxrss is in bytes, on Linux it is in KB size *= 1024 return size From 6aa0ba45532a4fd8e868bb4ea15bf002e430e67f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Tue, 3 Mar 2020 09:11:11 +0100 Subject: [PATCH 242/313] Write release notes for Scrapy 2.0.0 (#4329) --- docs/index.rst | 8 + docs/intro/install.rst | 4 +- docs/news.rst | 458 +++++++++++++++++++++++++- docs/topics/asyncio.rst | 28 ++ docs/topics/coroutines.rst | 110 +++++++ docs/topics/downloader-middleware.rst | 4 + docs/topics/exporters.rst | 5 +- docs/topics/feed-exports.rst | 3 + docs/topics/item-pipeline.rst | 13 +- docs/topics/jobs.rst | 3 + docs/topics/link-extractors.rst | 10 +- docs/topics/loaders.rst | 3 + docs/topics/media-pipeline.rst | 12 +- docs/topics/request-response.rst | 8 + docs/topics/settings.rst | 43 ++- docs/topics/signals.rst | 3 + docs/topics/spiders.rst | 3 + scrapy/http/response/__init__.py | 5 + scrapy/logformatter.py | 17 +- scrapy/utils/reactor.py | 5 + tests/test_crawl.py | 2 +- 21 files changed, 704 insertions(+), 43 deletions(-) create mode 100644 docs/topics/asyncio.rst create mode 100644 docs/topics/coroutines.rst diff --git a/docs/index.rst b/docs/index.rst index a4343b7e0..11aa5c9be 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -165,6 +165,8 @@ Solving specific problems topics/autothrottle topics/benchmarking topics/jobs + topics/coroutines + topics/asyncio :doc:`faq` Get answers to most frequently asked questions. @@ -205,6 +207,12 @@ Solving specific problems :doc:`topics/jobs` Learn how to pause and resume crawls for large spiders. +:doc:`topics/coroutines` + Use the :ref:`coroutine syntax <async>`. + +:doc:`topics/asyncio` + Use :mod:`asyncio` and :mod:`asyncio`-powered libraries. + .. _extending-scrapy: Extending Scrapy diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 89ba0c154..6356e0eea 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -7,8 +7,8 @@ Installation guide Installing Scrapy ================= -Scrapy runs on Python 3.5 or above -under CPython (default Python implementation) and PyPy (starting with PyPy 5.9). +Scrapy runs on Python 3.5 or above under CPython (default Python +implementation) and PyPy (starting with PyPy 5.9). If you're using `Anaconda`_ or `Miniconda`_, you can install the package from the `conda-forge`_ channel, which has up-to-date packages for Linux, Windows diff --git a/docs/news.rst b/docs/news.rst index c1daedaf2..dd5e00223 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -3,8 +3,452 @@ Release notes ============= -.. note:: Scrapy 1.x will be the last series supporting Python 2. Scrapy 2.0, - planned for Q4 2019 or Q1 2020, will support **Python 3 only**. +.. _release-2.0.0: + +Scrapy 2.0.0 (2020-03-03) +------------------------- + +Highlights: + +* Python 2 support has been removed +* :doc:`Partial <topics/coroutines>` :ref:`coroutine syntax <async>` support + and :doc:`experimental <topics/asyncio>` :mod:`asyncio` support +* New :meth:`Response.follow_all <scrapy.http.Response.follow_all>` method +* :ref:`FTP support <media-pipeline-ftp>` for media pipelines +* New :attr:`Response.certificate <scrapy.http.Response.certificate>` + attribute +* IPv6 support through :setting:`DNS_RESOLVER` + +Backward-incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* Python 2 support has been removed, following `Python 2 end-of-life on + January 1, 2020`_ (:issue:`4091`, :issue:`4114`, :issue:`4115`, + :issue:`4121`, :issue:`4138`, :issue:`4231`, :issue:`4242`, :issue:`4304`, + :issue:`4309`, :issue:`4373`) + +* Retry gaveups (see :setting:`RETRY_TIMES`) are now logged as errors instead + of as debug information (:issue:`3171`, :issue:`3566`) + +* File extensions that + :class:`LinkExtractor <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>` + ignores by default now also include ``7z``, ``7zip``, ``apk``, ``bz2``, + ``cdr``, ``dmg``, ``ico``, ``iso``, ``tar``, ``tar.gz``, ``webm``, and + ``xz`` (:issue:`1837`, :issue:`2067`, :issue:`4066`) + +* The :setting:`METAREFRESH_IGNORE_TAGS` setting is now an empty list by + default, following web browser behavior (:issue:`3844`, :issue:`4311`) + +* The + :class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware` + now includes spaces after commas in the value of the ``Accept-Encoding`` + header that it sets, following web browser behavior (:issue:`4293`) + +* The ``__init__`` method of custom download handlers (see + :setting:`DOWNLOAD_HANDLERS`) or subclasses of the following downloader + handlers no longer receives a ``settings`` parameter: + + * :class:`scrapy.core.downloader.handlers.datauri.DataURIDownloadHandler` + + * :class:`scrapy.core.downloader.handlers.file.FileDownloadHandler` + + Use the ``from_settings`` or ``from_crawler`` class methods to expose such + a parameter to your custom download handlers. + + (:issue:`4126`) + +* We have refactored the :class:`scrapy.core.scheduler.Scheduler` class and + related queue classes (see :setting:`SCHEDULER_PRIORITY_QUEUE`, + :setting:`SCHEDULER_DISK_QUEUE` and :setting:`SCHEDULER_MEMORY_QUEUE`) to + make it easier to implement custom scheduler queue classes. See + :ref:`2-0-0-scheduler-queue-changes` below for details. + +* Overridden settings are now logged in a different format. This is more in + line with similar information logged at startup (:issue:`4199`) + +.. _Python 2 end-of-life on January 1, 2020: https://www.python.org/doc/sunset-python-2/ + + +Deprecation removals +~~~~~~~~~~~~~~~~~~~~ + +* The :ref:`Scrapy shell <topics-shell>` no longer provides a `sel` proxy + object, use :meth:`response.selector <scrapy.http.Response.selector>` + instead (:issue:`4347`) + +* LevelDB support has been removed (:issue:`4112`) + +* The following functions have been removed from :mod:`scrapy.utils.python`: + ``isbinarytext``, ``is_writable``, ``setattr_default``, ``stringify_dict`` + (:issue:`4362`) + + +Deprecations +~~~~~~~~~~~~ + +* Using environment variables prefixed with ``SCRAPY_`` to override settings + is deprecated (:issue:`4300`, :issue:`4374`, :issue:`4375`) + +* :class:`scrapy.linkextractors.FilteringLinkExtractor` is deprecated, use + :class:`scrapy.linkextractors.LinkExtractor + <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>` instead (:issue:`4045`) + +* The ``noconnect`` query string argument of proxy URLs is deprecated and + should be removed from proxy URLs (:issue:`4198`) + +* The :meth:`next <scrapy.utils.python.MutableChain.next>` method of + :class:`scrapy.utils.python.MutableChain` is deprecated, use the global + :func:`next` function or :meth:`MutableChain.__next__ + <scrapy.utils.python.MutableChain.__next__>` instead (:issue:`4153`) + + +New features +~~~~~~~~~~~~ + +* Added :doc:`partial support <topics/coroutines>` for Python’s + :ref:`coroutine syntax <async>` and :doc:`experimental support + <topics/asyncio>` for :mod:`asyncio` and :mod:`asyncio`-powered libraries + (:issue:`4010`, :issue:`4259`, :issue:`4269`, :issue:`4270`, :issue:`4271`, + :issue:`4316`, :issue:`4318`) + +* The new :meth:`Response.follow_all <scrapy.http.Response.follow_all>` + method offers the same functionality as + :meth:`Response.follow <scrapy.http.Response.follow>` but supports an + iterable of URLs as input and returns an iterable of requests + (:issue:`2582`, :issue:`4057`, :issue:`4286`) + +* :ref:`Media pipelines <topics-media-pipeline>` now support :ref:`FTP + storage <media-pipeline-ftp>` (:issue:`3928`, :issue:`3961`) + +* The new :attr:`Response.certificate <scrapy.http.Response.certificate>` + attribute exposes the SSL certificate of the server as a + :class:`twisted.internet.ssl.Certificate` object for HTTPS responses + (:issue:`2726`, :issue:`4054`) + +* A new :setting:`DNS_RESOLVER` setting allows enabling IPv6 support + (:issue:`1031`, :issue:`4227`) + +* A new :setting:`SCRAPER_SLOT_MAX_ACTIVE_SIZE` setting allows configuring + the existing soft limit that pauses request downloads when the total + response data being processed is too high (:issue:`1410`, :issue:`3551`) + +* A new :setting:`TWISTED_REACTOR` setting allows customizing the + :mod:`~twisted.internet.reactor` that Scrapy uses, allowing to + :doc:`enable asyncio support <topics/asyncio>` or deal with a + :ref:`common macOS issue <faq-specific-reactor>` (:issue:`2905`, + :issue:`4294`) + +* Scheduler disk and memory queues may now use the class methods + ``from_crawler`` or ``from_settings`` (:issue:`3884`) + +* The new :attr:`Response.cb_kwargs <scrapy.http.Response.cb_kwargs>` + attribute serves as a shortcut for :attr:`Response.request.cb_kwargs + <scrapy.http.Request.cb_kwargs>` (:issue:`4331`) + +* :meth:`Response.follow <scrapy.http.Response.follow>` now supports a + ``flags`` parameter, for consistency with :class:`~scrapy.http.Request` + (:issue:`4277`, :issue:`4279`) + +* :ref:`Item loader processors <topics-loaders-processors>` can now be + regular functions, they no longer need to be methods (:issue:`3899`) + +* :class:`~scrapy.spiders.Rule` now accepts an ``errback`` parameter + (:issue:`4000`) + +* :class:`~scrapy.http.Request` no longer requires a ``callback`` parameter + when an ``errback`` parameter is specified (:issue:`3586`, :issue:`4008`) + +* :class:`~scrapy.logformatter.LogFormatter` now supports some additional + methods: + + * :class:`~scrapy.logformatter.LogFormatter.download_error` for + download errors + + * :class:`~scrapy.logformatter.LogFormatter.item_error` for exceptions + raised during item processing by :ref:`item pipelines + <topics-item-pipeline>` + + * :class:`~scrapy.logformatter.LogFormatter.spider_error` for exceptions + raised from :ref:`spider callbacks <topics-spiders>` + + (:issue:`374`, :issue:`3986`, :issue:`3989`, :issue:`4176`, :issue:`4188`) + +* The :setting:`FEED_URI` setting now supports :class:`pathlib.Path` values + (:issue:`3731`, :issue:`4074`) + +* A new :signal:`request_left_downloader` signal is sent when a request + leaves the downloader (:issue:`4303`) + +* Scrapy logs a warning when it detects a request callback or errback that + uses ``yield`` but also returns a value, since the returned value would be + lost (:issue:`3484`, :issue:`3869`) + +* :class:`~scrapy.spiders.Spider` objects now raise an :exc:`AttributeError` + exception if they do not have a :class:`~scrapy.spiders.Spider.start_urls` + attribute nor reimplement :class:`~scrapy.spiders.Spider.start_requests`, + but have a ``start_url`` attribute (:issue:`4133`, :issue:`4170`) + +* :class:`~scrapy.exporters.BaseItemExporter` subclasses may now use + ``super().__init__(**kwargs)`` instead of ``self._configure(kwargs)`` in + their ``__init__`` method, passing ``dont_fail=True`` to the parent + ``__init__`` method if needed, and accessing ``kwargs`` at ``self._kwargs`` + after calling their parent ``__init__`` method (:issue:`4193`, + :issue:`4370`) + +* A new ``keep_fragments`` parameter of + :func:`scrapy.utils.request.request_fingerprint` allows to generate + different fingerprints for requests with different fragments in their URL + (:issue:`4104`) + +* Download handlers (see :setting:`DOWNLOAD_HANDLERS`) may now use the + ``from_settings`` and ``from_crawler`` class methods that other Scrapy + components already supported (:issue:`4126`) + +* :class:`scrapy.utils.python.MutableChain.__iter__` now returns ``self``, + `allowing it to be used as a sequence <https://lgtm.com/rules/4850080/>`_ + (:issue:`4153`) + + +Bug fixes +~~~~~~~~~ + +* The :command:`crawl` command now also exits with exit code 1 when an + exception happens before the crawling starts (:issue:`4175`, :issue:`4207`) + +* :class:`LinkExtractor.extract_links + <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` no longer + re-encodes the query string or URLs from non-UTF-8 responses in UTF-8 + (:issue:`998`, :issue:`1403`, :issue:`1949`, :issue:`4321`) + +* The first spider middleware (see :setting:`SPIDER_MIDDLEWARES`) now also + processes exceptions raised from callbacks that are generators + (:issue:`4260`, :issue:`4272`) + +* Redirects to URLs starting with 3 slashes (``///``) are now supported + (:issue:`4032`, :issue:`4042`) + +* :class:`~scrapy.http.Request` no longer accepts strings as ``url`` simply + because they have a colon (:issue:`2552`, :issue:`4094`) + +* The correct encoding is now used for attach names in + :class:`~scrapy.mail.MailSender` (:issue:`4229`, :issue:`4239`) + +* :class:`~scrapy.dupefilters.RFPDupeFilter`, the default + :setting:`DUPEFILTER_CLASS`, no longer writes an extra ``\r`` character on + each line in Windows, which made the size of the ``requests.seen`` file + unnecessarily large on that platform (:issue:`4283`) + +* Z shell auto-completion now looks for ``.html`` files, not ``.http`` files, + and covers the ``-h`` command-line switch (:issue:`4122`, :issue:`4291`) + +* Adding items to a :class:`scrapy.utils.datatypes.LocalCache` object + without a ``limit`` defined no longer raises a :exc:`TypeError` exception + (:issue:`4123`) + +* Fixed a typo in the message of the :exc:`ValueError` exception raised when + :func:`scrapy.utils.misc.create_instance` gets both ``settings`` and + ``crawler`` set to ``None`` (:issue:`4128`) + + +Documentation +~~~~~~~~~~~~~ + +* API documentation now links to an online, syntax-highlighted view of the + corresponding source code (:issue:`4148`) + +* Links to unexisting documentation pages now allow access to the sidebar + (:issue:`4152`, :issue:`4169`) + +* Cross-references within our documentation now display a tooltip when + hovered (:issue:`4173`, :issue:`4183`) + +* Improved the documentation about :meth:`LinkExtractor.extract_links + <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` and + simplified :ref:`topics-link-extractors` (:issue:`4045`) + +* Clarified how :class:`ItemLoader.item <scrapy.loader.ItemLoader.item>` + works (:issue:`3574`, :issue:`4099`) + +* Clarified that :func:`logging.basicConfig` should not be used when also + using :class:`~scrapy.crawler.CrawlerProcess` (:issue:`2149`, + :issue:`2352`, :issue:`3146`, :issue:`3960`) + +* Clarified the requirements for :class:`~scrapy.http.Request` objects + :ref:`when using persistence <request-serialization>` (:issue:`4124`, + :issue:`4139`) + +* Clarified how to install a :ref:`custom image pipeline + <media-pipeline-example>` (:issue:`4034`, :issue:`4252`) + +* Fixed the signatures of the ``file_path`` method in :ref:`media pipeline + <topics-media-pipeline>` examples (:issue:`4290`) + +* Covered a backward-incompatible change in Scrapy 1.7.0 affecting custom + :class:`scrapy.core.scheduler.Scheduler` subclasses (:issue:`4274`) + +* Improved the ``README.rst`` and ``CODE_OF_CONDUCT.md`` files + (:issue:`4059`) + +* Documentation examples are now checked as part of our test suite and we + have fixed some of the issues detected (:issue:`4142`, :issue:`4146`, + :issue:`4171`, :issue:`4184`, :issue:`4190`) + +* Fixed logic issues, broken links and typos (:issue:`4247`, :issue:`4258`, + :issue:`4282`, :issue:`4288`, :issue:`4305`, :issue:`4308`, :issue:`4323`, + :issue:`4338`, :issue:`4359`, :issue:`4361`) + +* Improved consistency when referring to the ``__init__`` method of an object + (:issue:`4086`, :issue:`4088`) + +* Fixed an inconsistency between code and output in :ref:`intro-overview` + (:issue:`4213`) + +* Extended :mod:`~sphinx.ext.intersphinx` usage (:issue:`4147`, + :issue:`4172`, :issue:`4185`, :issue:`4194`, :issue:`4197`) + +* We now use a recent version of Python to build the documentation + (:issue:`4140`, :issue:`4249`) + +* Cleaned up documentation (:issue:`4143`, :issue:`4275`) + + +Quality assurance +~~~~~~~~~~~~~~~~~ + +* Re-enabled proxy ``CONNECT`` tests (:issue:`2545`, :issue:`4114`) + +* Added Bandit_ security checks to our test suite (:issue:`4162`, + :issue:`4181`) + +* Added Flake8_ style checks to our test suite and applied many of the + corresponding changes (:issue:`3944`, :issue:`3945`, :issue:`4137`, + :issue:`4157`, :issue:`4167`, :issue:`4174`, :issue:`4186`, :issue:`4195`, + :issue:`4238`, :issue:`4246`, :issue:`4355`, :issue:`4360`, :issue:`4365`) + +* Improved test coverage (:issue:`4097`, :issue:`4218`, :issue:`4236`) + +* Started reporting slowest tests, and improved the performance of some of + them (:issue:`4163`, :issue:`4164`) + +* Fixed broken tests and refactored some tests (:issue:`4014`, :issue:`4095`, + :issue:`4244`, :issue:`4268`, :issue:`4372`) + +* Modified the :doc:`tox <tox:index>` configuration to allow running tests + with any Python version, run Bandit_ and Flake8_ tests by default, and + enforce a minimum tox version programmatically (:issue:`4179`) + +* Cleaned up code (:issue:`3937`, :issue:`4208`, :issue:`4209`, + :issue:`4210`, :issue:`4212`, :issue:`4369`, :issue:`4376`, :issue:`4378`) + +.. _Bandit: https://bandit.readthedocs.io/ +.. _Flake8: https://flake8.pycqa.org/en/latest/ + + +.. _2-0-0-scheduler-queue-changes: + +Changes to scheduler queue classes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +The following changes may impact any custom queue classes of all types: + +* The ``push`` method no longer receives a second positional parameter + containing ``request.priority * -1``. If you need that value, get it + from the first positional parameter, ``request``, instead, or use + the new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority` + method in :class:`scrapy.core.scheduler.ScrapyPriorityQueue` + subclasses. + +The following changes may impact custom priority queue classes: + +* In the ``__init__`` method or the ``from_crawler`` or ``from_settings`` + class methods: + + * The parameter that used to contain a factory function, + ``qfactory``, is now passed as a keyword parameter named + ``downstream_queue_cls``. + + * A new keyword parameter has been added: ``key``. It is a string + that is always an empty string for memory queues and indicates the + :setting:`JOB_DIR` value for disk queues. + + * The parameter for disk queues that contains data from the previous + crawl, ``startprios`` or ``slot_startprios``, is now passed as a + keyword parameter named ``startprios``. + + * The ``serialize`` parameter is no longer passed. The disk queue + class must take care of request serialization on its own before + writing to disk, using the + :func:`~scrapy.utils.reqser.request_to_dict` and + :func:`~scrapy.utils.reqser.request_from_dict` functions from the + :mod:`scrapy.utils.reqser` module. + +The following changes may impact custom disk and memory queue classes: + +* The signature of the ``__init__`` method is now + ``__init__(self, crawler, key)``. + +The following changes affect specifically the +:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` and +:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` classes from +:mod:`scrapy.core.scheduler` and may affect subclasses: + +* In the ``__init__`` method, most of the changes described above apply. + + ``__init__`` may still receive all parameters as positional parameters, + however: + + * ``downstream_queue_cls``, which replaced ``qfactory``, must be + instantiated differently. + + ``qfactory`` was instantiated with a priority value (integer). + + Instances of ``downstream_queue_cls`` should be created using + the new + :meth:`ScrapyPriorityQueue.qfactory <scrapy.core.scheduler.ScrapyPriorityQueue.qfactory>` + or + :meth:`DownloaderAwarePriorityQueue.pqfactory <scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqfactory>` + methods. + + * The new ``key`` parameter displaced the ``startprios`` + parameter 1 position to the right. + +* The following class attributes have been added: + + * :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.crawler` + + * :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.downstream_queue_cls` + (details above) + + * :attr:`~scrapy.core.scheduler.ScrapyPriorityQueue.key` (details above) + +* The ``serialize`` attribute has been removed (details above) + +The following changes affect specifically the +:class:`~scrapy.core.scheduler.ScrapyPriorityQueue` class and may affect +subclasses: + +* A new :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.priority` + method has been added which, given a request, returns + ``request.priority * -1``. + + It is used in :meth:`~scrapy.core.scheduler.ScrapyPriorityQueue.push` + to make up for the removal of its ``priority`` parameter. + +* The ``spider`` attribute has been removed. Use + :attr:`crawler.spider <scrapy.core.scheduler.ScrapyPriorityQueue.crawler>` + instead. + +The following changes affect specifically the +:class:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue` class and may +affect subclasses: + +* A new :attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.pqueues` + attribute offers a mapping of downloader slot names to the + corresponding instances of + :attr:`~scrapy.core.scheduler.DownloaderAwarePriorityQueue.downstream_queue_cls`. + +(:issue:`3884`) + .. _release-1.8.0: @@ -288,12 +732,12 @@ Backward-incompatible changes :class:`~scrapy.http.Request` objects instead of arbitrary Python data structures. -* An additional ``crawler`` parameter has been added to the ``__init__`` method - of the :class:`scrapy.core.scheduler.Scheduler` class. - Custom scheduler subclasses which don't accept arbitrary parameters in - their ``__init__`` method might break because of this change. +* An additional ``crawler`` parameter has been added to the ``__init__`` + method of the :class:`~scrapy.core.scheduler.Scheduler` class. Custom + scheduler subclasses which don't accept arbitrary parameters in their + ``__init__`` method might break because of this change. - For more information, refer to the documentation for the :setting:`SCHEDULER` setting. + For more information, see :setting:`SCHEDULER`. See also :ref:`1.7-deprecation-removals` below. diff --git a/docs/topics/asyncio.rst b/docs/topics/asyncio.rst new file mode 100644 index 000000000..038a459fd --- /dev/null +++ b/docs/topics/asyncio.rst @@ -0,0 +1,28 @@ +======= +asyncio +======= + +.. versionadded:: 2.0 + +Scrapy has partial support :mod:`asyncio`. After you :ref:`install the asyncio +reactor <install-asyncio>`, you may use :mod:`asyncio` and +:mod:`asyncio`-powered libraries in any :doc:`coroutine <coroutines>`. + +.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy + versions may introduce related changes without a deprecation + period or warning. + +.. _install-asyncio: + +Installing the asyncio reactor +============================== + +To enable :mod:`asyncio` support, set the :setting:`TWISTED_REACTOR` setting to +``'twisted.internet.asyncioreactor.AsyncioSelectorReactor'``. + +If you are using :class:`~scrapy.crawler.CrawlerRunner`, you also need to +install the :class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor` +reactor manually. You can do that using +:func:`~scrapy.utils.reactor.install_reactor`:: + + install_reactor('twisted.internet.asyncioreactor.AsyncioSelectorReactor') diff --git a/docs/topics/coroutines.rst b/docs/topics/coroutines.rst new file mode 100644 index 000000000..487cf4c6c --- /dev/null +++ b/docs/topics/coroutines.rst @@ -0,0 +1,110 @@ +========== +Coroutines +========== + +.. versionadded:: 2.0 + +Scrapy has :ref:`partial support <coroutine-support>` for the +:ref:`coroutine syntax <async>`. + +.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy + versions may introduce related API and behavior changes without a + deprecation period or warning. + +.. _coroutine-support: + +Supported callables +=================== + +The following callables may be defined as coroutines using ``async def``, and +hence use coroutine syntax (e.g. ``await``, ``async for``, ``async with``): + +- :class:`~scrapy.http.Request` callbacks. + + The following are known caveats of the current implementation that we aim + to address in future versions of Scrapy: + + - The callback output is not processed until the whole callback finishes. + + As a side effect, if the callback raises an exception, none of its + output is processed. + + - Because `asynchronous generators were introduced in Python 3.6`_, you + can only use ``yield`` if you are using Python 3.6 or later. + + If you need to output multiple items or requests and you are using + Python 3.5, return an iterable (e.g. a list) instead. + +- The :meth:`process_item` method of + :ref:`item pipelines <topics-item-pipeline>`. + +- The + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_request`, + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_response`, + and + :meth:`~scrapy.downloadermiddlewares.DownloaderMiddleware.process_exception` + methods of + :ref:`downloader middlewares <topics-downloader-middleware-custom>`. + +- :ref:`Signal handlers that support deferreds <signal-deferred>`. + +.. _asynchronous generators were introduced in Python 3.6: https://www.python.org/dev/peps/pep-0525/ + +Usage +===== + +There are several use cases for coroutines in Scrapy. Code that would +return Deferreds when written for previous Scrapy versions, such as downloader +middlewares and signal handlers, can be rewritten to be shorter and cleaner:: + + class DbPipeline: + def _update_item(self, data, item): + item['field'] = data + return item + + def process_item(self, item, spider): + dfd = db.get_some_data(item['id']) + dfd.addCallback(self._update_item, item) + return dfd + +becomes:: + + class DbPipeline: + async def process_item(self, item, spider): + item['field'] = await db.get_some_data(item['id']) + return item + +Coroutines may be used to call asynchronous code. This includes other +coroutines, functions that return Deferreds and functions that return +`awaitable objects`_ such as :class:`~asyncio.Future`. This means you can use +many useful Python libraries providing such code:: + + class MySpider(Spider): + # ... + async def parse_with_deferred(self, response): + additional_response = await treq.get('https://additional.url') + additional_data = await treq.content(additional_response) + # ... use response and additional_data to yield items and requests + + async def parse_with_asyncio(self, response): + async with aiohttp.ClientSession() as session: + async with session.get('https://additional.url') as additional_response: + additional_data = await r.text() + # ... use response and additional_data to yield items and requests + +.. note:: Many libraries that use coroutines, such as `aio-libs`_, require the + :mod:`asyncio` loop and to use them you need to + :doc:`enable asyncio support in Scrapy<asyncio>`. + +Common use cases for asynchronous code include: + +* requesting data from websites, databases and other services (in callbacks, + pipelines and middlewares); +* storing data in databases (in pipelines and middlewares); +* delaying the spider initialization until some external event (in the + :signal:`spider_opened` handler); +* calling asynchronous Scrapy methods like ``ExecutionEngine.download`` (see + :ref:`the screenshot pipeline example<ScreenshotPipeline>`). + +.. _aio-libs: https://github.com/aio-libs +.. _awaitable objects: https://docs.python.org/3/glossary.html#term-awaitable diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 0297ef3a0..73648994d 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -872,6 +872,10 @@ Default: ``[]`` Meta tags within these tags are ignored. +.. versionchanged:: 2.0 + The default value of :setting:`METAREFRESH_IGNORE_TAGS` changed from + ``['script', 'noscript']`` to ``[]``. + .. setting:: METAREFRESH_MAXDELAY METAREFRESH_MAXDELAY diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index b8d898022..d411e2eed 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -137,7 +137,7 @@ output examples, which assume you're exporting these two items:: BaseItemExporter ---------------- -.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0) +.. class:: BaseItemExporter(fields_to_export=None, export_empty_fields=False, encoding='utf-8', indent=0, dont_fail=False) This is the (abstract) base class for all Item Exporters. It provides support for common features used by all (concrete) Item Exporters, such as @@ -148,6 +148,9 @@ BaseItemExporter populate their respective instance attributes: :attr:`fields_to_export`, :attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`. + .. versionadded:: 2.0 + The *dont_fail* parameter. + .. method:: export_item(item) Exports the given item. This method must be implemented in subclasses. diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index 1d94807a4..42f1cad90 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -236,6 +236,9 @@ supported URI schemes. This setting is required for enabling the feed exports. +.. versionchanged:: 2.0 + Added :class:`pathlib.Path` support. + .. setting:: FEED_FORMAT FEED_FORMAT diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index 801d48fd5..98e2506e5 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -162,14 +162,16 @@ method and how to clean up the resources properly.:: .. _pymongo: https://api.mongodb.com/python/current/ +.. _ScreenshotPipeline: + Take screenshot of item ----------------------- This example demonstrates how to return a :class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method. It uses Splash_ to render screenshot of item url. Pipeline -makes request to locally running instance of Splash_. After request is downloaded -and Deferred callback fires, it saves item to a file and adds filename to an item. +makes request to locally running instance of Splash_. After request is downloaded, +it saves the screenshot to a file and adds filename to the item. :: @@ -184,15 +186,12 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite SPLASH_URL = "http://localhost:8050/render.png?url={}" - def process_item(self, item, spider): + async def process_item(self, item, spider): encoded_item_url = quote(item["url"]) screenshot_url = self.SPLASH_URL.format(encoded_item_url) request = scrapy.Request(screenshot_url) - dfd = spider.crawler.engine.download(request, spider) - dfd.addBoth(self.return_item, item) - return dfd + response = await spider.crawler.engine.download(request, spider) - def return_item(self, response, item): if response.status != 200: # Error happened, return item. return item diff --git a/docs/topics/jobs.rst b/docs/topics/jobs.rst index c34ba336b..58601824a 100644 --- a/docs/topics/jobs.rst +++ b/docs/topics/jobs.rst @@ -68,6 +68,9 @@ Cookies may expire. So, if you don't resume your spider quickly the requests scheduled may no longer work. This won't be an issue if you spider doesn't rely on cookies. + +.. _request-serialization: + Request serialization --------------------- diff --git a/docs/topics/link-extractors.rst b/docs/topics/link-extractors.rst index 8c8019438..0162a331a 100644 --- a/docs/topics/link-extractors.rst +++ b/docs/topics/link-extractors.rst @@ -64,9 +64,13 @@ LxmlLinkExtractor :param deny_extensions: a single value or list of strings containing extensions that should be ignored when extracting links. - If not given, it will default to the - ``IGNORED_EXTENSIONS`` list defined in the - `scrapy.linkextractors`_ package. + If not given, it will default to + :data:`scrapy.linkextractors.IGNORED_EXTENSIONS`. + + .. versionchanged:: 2.0 + :data:`~scrapy.linkextractors.IGNORED_EXTENSIONS` now includes + ``7z``, ``7zip``, ``apk``, ``bz2``, ``cdr``, ``dmg``, ``ico``, + ``iso``, ``tar``, ``tar.gz``, ``webm``, and ``xz``. :type deny_extensions: list :param restrict_xpaths: is an XPath (or list of XPath's) which defines diff --git a/docs/topics/loaders.rst b/docs/topics/loaders.rst index 9d5fccbbc..5f75ccbff 100644 --- a/docs/topics/loaders.rst +++ b/docs/topics/loaders.rst @@ -136,6 +136,9 @@ with the data to be parsed, and return a parsed value. So you can use any function as input or output processor. The only requirement is that they must accept one (and only one) positional argument, which will be an iterable. +.. versionchanged:: 2.0 + Processors no longer need to be methods. + .. note:: Both input and output processors must receive an iterable as their first argument. The output of those functions can be anything. The result of input processors will be appended to an internal list (in the Loader) diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index 67a0bfdba..cd84905c5 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -116,12 +116,6 @@ For the Images Pipeline, set the :setting:`IMAGES_STORE` setting:: Supported Storage ================= -File system is currently the only officially supported storage, but there are -also support for storing files in `Amazon S3`_ and `Google Cloud Storage`_. - -.. _Amazon S3: https://aws.amazon.com/s3/ -.. _Google Cloud Storage: https://cloud.google.com/storage/ - File system storage ------------------- @@ -147,9 +141,13 @@ Where: * ``full`` is a sub-directory to separate full images from thumbnails (if used). For more info see :ref:`topics-images-thumbnails`. +.. _media-pipeline-ftp: + FTP server storage ------------------ +.. versionadded:: 2.0 + :setting:`FILES_STORE` and :setting:`IMAGES_STORE` can point to an FTP server. Scrapy will automatically upload the files to the server. @@ -573,6 +571,8 @@ See here the methods that you can override in your custom Images Pipeline: By default, the :meth:`item_completed` method returns the item. +.. _media-pipeline-example: + Custom Images pipeline example ============================== diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index c4c2845c9..b2a60ff39 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -31,6 +31,8 @@ Request objects a :class:`Response`. :param url: the URL of this request + + If the URL is invalid, a :exc:`ValueError` exception is raised. :type url: string :param callback: the function that will be called with the response of this @@ -125,6 +127,10 @@ Request objects :exc:`~twisted.python.failure.Failure` as first parameter. For more information, see :ref:`topics-request-response-ref-errbacks` below. + + .. versionchanged:: 2.0 + The *callback* parameter is no longer required when the *errback* + parameter is specified. :type errback: callable :param flags: Flags sent to the request, can be used for logging or similar purposes. @@ -677,6 +683,8 @@ Response objects .. attribute:: Response.cb_kwargs + .. versionadded:: 2.0 + A shortcut to the :attr:`Request.cb_kwargs` attribute of the :attr:`Response.request` object (i.e. ``self.request.cb_kwargs``). diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 5394147da..a70023efa 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -381,6 +381,8 @@ DNS in-memory cache size. DNS_RESOLVER ------------ +.. versionadded:: 2.0 + Default: ``'scrapy.resolver.CachingThreadedResolver'`` The class to be used to resolve DNS names. The default ``scrapy.resolver.CachingThreadedResolver`` @@ -1258,6 +1260,9 @@ does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`. SCRAPER_SLOT_MAX_ACTIVE_SIZE ---------------------------- + +.. versionadded:: 2.0 + Default: ``5_000_000`` Soft limit (in bytes) for response data being processed. @@ -1447,24 +1452,36 @@ in the ``project`` subdirectory. TWISTED_REACTOR --------------- +.. versionadded:: 2.0 + Default: ``None`` -Import path of a given Twisted reactor, for instance: -:class:`twisted.internet.asyncioreactor.AsyncioSelectorReactor`. +Import path of a given :mod:`~twisted.internet.reactor`. -Scrapy will install this reactor if no other is installed yet, such as when -the ``scrapy`` CLI program is invoked or when using the -:class:`~scrapy.crawler.CrawlerProcess` class. If you are using the -:class:`~scrapy.crawler.CrawlerRunner` class, you need to install the correct -reactor manually. An exception will be raised if the installation fails. +Scrapy will install this reactor if no other reactor is installed yet, such as +when the ``scrapy`` CLI program is invoked or when using the +:class:`~scrapy.crawler.CrawlerProcess` class. -The default value for this option is currently ``None``, which means that Scrapy -will not attempt to install any specific reactor, and the default one defined by -Twisted for the current platform will be used. This is to maintain backward -compatibility and avoid possible problems caused by using a non-default reactor. +If you are using the :class:`~scrapy.crawler.CrawlerRunner` class, you also +need to install the correct reactor manually. You can do that using +:func:`~scrapy.utils.reactor.install_reactor`: -For additional information, please see -:doc:`core/howto/choosing-reactor`. +.. autofunction:: scrapy.utils.reactor.install_reactor + +If a reactor is already installed, +:func:`~scrapy.utils.reactor.install_reactor` has no effect. + +:meth:`CrawlerRunner.__init__ <scrapy.crawler.CrawlerRunner.__init__>` raises +:exc:`Exception` if the installed reactor does not match the +:setting:`TWISTED_REACTOR` setting. + +The default value of the :setting:`TWISTED_REACTOR` setting is ``None``, which +means that Scrapy will not attempt to install any specific reactor, and the +default reactor defined by Twisted for the current platform will be used. This +is to maintain backward compatibility and avoid possible problems caused by +using a non-default reactor. + +For additional information, see :doc:`core/howto/choosing-reactor`. .. setting:: URLLENGTH_LIMIT diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index d3cfb0307..2def53848 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -46,6 +46,7 @@ Here is a simple example showing how you can catch signals and perform some acti def parse(self, response): pass +.. _signal-deferred: Deferred signal handlers ======================== @@ -301,6 +302,8 @@ request_left_downloader .. signal:: request_left_downloader .. function:: request_left_downloader(request, spider) + .. versionadded:: 2.0 + Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of failure. diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index e0f33de66..89609db7d 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -420,6 +420,9 @@ Crawling rules It receives a :class:`Twisted Failure <twisted.python.failure.Failure>` instance as first parameter. + .. versionadded:: 2.0 + The *errback* parameter. + CrawlSpider example ~~~~~~~~~~~~~~~~~~~ diff --git a/scrapy/http/response/__init__.py b/scrapy/http/response/__init__.py index 119dd2f63..682cec161 100644 --- a/scrapy/http/response/__init__.py +++ b/scrapy/http/response/__init__.py @@ -129,6 +129,9 @@ class Response(object_ref): :class:`~.TextResponse` provides a :meth:`~.TextResponse.follow` method which supports selectors in addition to absolute/relative URLs and Link objects. + + .. versionadded:: 2.0 + The *flags* parameter. """ if isinstance(url, Link): url = url.url @@ -157,6 +160,8 @@ class Response(object_ref): dont_filter=False, errback=None, cb_kwargs=None, flags=None): # type: (...) -> Generator[Request, None, None] """ + .. versionadded:: 2.0 + Return an iterable of :class:`~.Request` instances to follow all links in ``urls``. It accepts the same arguments as ``Request.__init__`` method, but elements of ``urls`` can be relative URLs or :class:`~scrapy.link.Link` objects, diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 194013642..14cec44a6 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -97,7 +97,11 @@ class LogFormatter(object): } def item_error(self, item, exception, response, spider): - """Logs a message when an item causes an error while it is passing through the item pipeline.""" + """Logs a message when an item causes an error while it is passing + through the item pipeline. + + .. versionadded:: 2.0 + """ return { 'level': logging.ERROR, 'msg': ITEMERRORMSG, @@ -107,7 +111,10 @@ class LogFormatter(object): } def spider_error(self, failure, request, response, spider): - """Logs an error message from a spider.""" + """Logs an error message from a spider. + + .. versionadded:: 2.0 + """ return { 'level': logging.ERROR, 'msg': SPIDERERRORMSG, @@ -118,7 +125,11 @@ class LogFormatter(object): } def download_error(self, failure, request, spider, errmsg=None): - """Logs a download error message from a spider (typically coming from the engine).""" + """Logs a download error message from a spider (typically coming from + the engine). + + .. versionadded:: 2.0 + """ args = {'request': request} if errmsg: msg = DOWNLOADERRORMSG_LONG diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 6513e06c9..17d6b2857 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -50,6 +50,8 @@ class CallLaterOnce(object): def install_reactor(reactor_path): + """Installs the :mod:`~twisted.internet.reactor` with the specified + import path.""" reactor_class = load_object(reactor_path) if reactor_class is asyncioreactor.AsyncioSelectorReactor: with suppress(error.ReactorAlreadyInstalledError): @@ -63,6 +65,9 @@ def install_reactor(reactor_path): def verify_installed_reactor(reactor_path): + """Raises :exc:`Exception` if the installed + :mod:`~twisted.internet.reactor` does not match the specified import + path.""" from twisted.internet import reactor reactor_class = load_object(reactor_path) if not isinstance(reactor, reactor_class): diff --git a/tests/test_crawl.py b/tests/test_crawl.py index e93c668c5..3f8a7435c 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -325,7 +325,7 @@ with multiples lines @mark.only_asyncio() @defer.inlineCallbacks def test_async_def_asyncio_parse(self): - runner = CrawlerRunner({"ASYNCIO_REACTOR": True}) + runner = CrawlerRunner({"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor"}) runner.crawl(AsyncDefAsyncioSpider, self.mockserver.url("/status?n=200"), mockserver=self.mockserver) with LogCapture() as log: yield runner.join() From a4dbb7754b999c8c6a5239bb3f58e951369e017e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Tue, 3 Mar 2020 09:13:00 +0100 Subject: [PATCH 243/313] =?UTF-8?q?Bump=20version:=201.8.0=20=E2=86=92=202?= =?UTF-8?q?.0.0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.cfg | 3 +-- scrapy/VERSION | 2 +- 2 files changed, 2 insertions(+), 3 deletions(-) diff --git a/.bumpversion.cfg b/.bumpversion.cfg index c9f1abea5..f347a0cd0 100644 --- a/.bumpversion.cfg +++ b/.bumpversion.cfg @@ -1,8 +1,7 @@ [bumpversion] -current_version = 1.8.0 +current_version = 2.0.0 commit = True tag = True tag_name = {new_version} [bumpversion:file:scrapy/VERSION] - diff --git a/scrapy/VERSION b/scrapy/VERSION index 27f9cd322..227cea215 100644 --- a/scrapy/VERSION +++ b/scrapy/VERSION @@ -1 +1 @@ -1.8.0 +2.0.0 From 1b591ff061f2b38bf328e1d2a4acd9643d45ad80 Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Fri, 28 Feb 2020 02:10:13 +0000 Subject: [PATCH 244/313] Obsolete deprecated settings Obsolete REDIRECT_MAX_METAREFRESH_DELAY which has been deprecated since Scrapy 0.18 Obsolete LOG_UNSERIALIZABLE_REQUESTS which has been deprecated since Scrapy 1.2.0 and is replaced by SCHEDULER_DEBUG --- scrapy/core/scheduler.py | 3 +-- scrapy/downloadermiddlewares/redirect.py | 3 +-- scrapy/settings/deprecated.py | 2 -- tests/test_scheduler.py | 2 +- 4 files changed, 3 insertions(+), 7 deletions(-) diff --git a/scrapy/core/scheduler.py b/scrapy/core/scheduler.py index e184ed50e..c96b9b719 100644 --- a/scrapy/core/scheduler.py +++ b/scrapy/core/scheduler.py @@ -66,8 +66,7 @@ class Scheduler(object): dqclass = load_object(settings['SCHEDULER_DISK_QUEUE']) mqclass = load_object(settings['SCHEDULER_MEMORY_QUEUE']) - logunser = settings.getbool('LOG_UNSERIALIZABLE_REQUESTS', - settings.getbool('SCHEDULER_DEBUG')) + logunser = settings.getbool('SCHEDULER_DEBUG') return cls(dupefilter, jobdir=job_dir(settings), logunser=logunser, stats=crawler.stats, pqclass=pqclass, dqclass=dqclass, mqclass=mqclass, crawler=crawler) diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 77cb5aa94..08cff8a55 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -93,8 +93,7 @@ class MetaRefreshMiddleware(BaseRedirectMiddleware): def __init__(self, settings): super(MetaRefreshMiddleware, self).__init__(settings) self._ignore_tags = settings.getlist('METAREFRESH_IGNORE_TAGS') - self._maxdelay = settings.getint('REDIRECT_MAX_METAREFRESH_DELAY', - settings.getint('METAREFRESH_MAXDELAY')) + self._maxdelay = settings.getint('METAREFRESH_MAXDELAY') def process_response(self, request, response, spider): if request.meta.get('dont_redirect', False) or request.method == 'HEAD' or \ diff --git a/scrapy/settings/deprecated.py b/scrapy/settings/deprecated.py index 1211908df..f6f878725 100644 --- a/scrapy/settings/deprecated.py +++ b/scrapy/settings/deprecated.py @@ -11,8 +11,6 @@ DEPRECATED_SETTINGS = [ ('SQLITE_DB', 'no longer supported'), ('AUTOTHROTTLE_MIN_DOWNLOAD_DELAY', 'use DOWNLOAD_DELAY instead'), ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), - ('REDIRECT_MAX_METAREFRESH_DELAY', 'use METAREFRESH_MAXDELAY instead'), - ('LOG_UNSERIALIZABLE_REQUESTS', 'use SCHEDULER_DEBUG instead'), ] diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index e0e3600e5..13c297084 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -46,7 +46,7 @@ class MockCrawler(Crawler): def __init__(self, priority_queue_cls, jobdir): settings = dict( - LOG_UNSERIALIZABLE_REQUESTS=False, + SCHEDULER_DEBUG=False, SCHEDULER_DISK_QUEUE='scrapy.squeues.PickleLifoDiskQueue', SCHEDULER_MEMORY_QUEUE='scrapy.squeues.LifoMemoryQueue', SCHEDULER_PRIORITY_QUEUE=priority_queue_cls, From 64002255554aa2aa79863b657e5d1674d72b228d Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Fri, 28 Feb 2020 00:06:11 +0000 Subject: [PATCH 245/313] Drop horribly outdated deb package build files --- Makefile.buildbot | 24 -------------------- debian/changelog | 5 ----- debian/compat | 1 - debian/control | 20 ----------------- debian/copyright | 40 --------------------------------- debian/pyversions | 1 - debian/rules | 5 ----- debian/scrapy.docs | 2 -- debian/scrapy.install | 2 -- debian/scrapy.lintian-overrides | 1 - debian/scrapy.manpages | 1 - 11 files changed, 102 deletions(-) delete mode 100644 Makefile.buildbot delete mode 100644 debian/changelog delete mode 100644 debian/compat delete mode 100644 debian/control delete mode 100644 debian/copyright delete mode 100644 debian/pyversions delete mode 100755 debian/rules delete mode 100644 debian/scrapy.docs delete mode 100644 debian/scrapy.install delete mode 100644 debian/scrapy.lintian-overrides delete mode 100644 debian/scrapy.manpages diff --git a/Makefile.buildbot b/Makefile.buildbot deleted file mode 100644 index 775538259..000000000 --- a/Makefile.buildbot +++ /dev/null @@ -1,24 +0,0 @@ -TRIAL := $(shell which trial) -BRANCH := $(shell git rev-parse --abbrev-ref HEAD) -export PYTHONPATH=$(PWD) - -test: - coverage run --branch $(TRIAL) --reporter=text tests - rm -rf htmlcov && coverage html - -s3cmd sync -P htmlcov/ s3://static.scrapy.org/coverage-scrapy-$(BRANCH)/ - -build: - git describe --tags --match '[0-9]*' |sed 's/-/.post/;s/-g/+g/' >scrapy/VERSION - debchange -m -D unstable --force-distribution -v \ - $$(python setup.py --version |sed -r 's/([0-9]+.[0-9]+.[0-9]+)(a|b|rc|dev)([0-9]*)/\1~\2\3/')-$$(date +%s) \ - "Automatic build" - debuild -us -uc -b - -clean: - git checkout debian scrapy/VERSION - git clean -dfq - -pypi: - umask 0022 && chmod -R a+rX . && python setup.py sdist upload - -.PHONY: clean test build diff --git a/debian/changelog b/debian/changelog deleted file mode 100644 index dde97f9e3..000000000 --- a/debian/changelog +++ /dev/null @@ -1,5 +0,0 @@ -scrapy (0.11) unstable; urgency=low - - * Initial release. - - -- Scrapinghub Team <info@scrapinghub.com> Thu, 10 Jun 2010 17:24:02 -0300 diff --git a/debian/compat b/debian/compat deleted file mode 100644 index 7f8f011eb..000000000 --- a/debian/compat +++ /dev/null @@ -1 +0,0 @@ -7 diff --git a/debian/control b/debian/control deleted file mode 100644 index 2cc8eedf4..000000000 --- a/debian/control +++ /dev/null @@ -1,20 +0,0 @@ -Source: scrapy -Section: python -Priority: optional -Maintainer: Scrapinghub Team <info@scrapinghub.com> -Build-Depends: debhelper (>= 7.0.50), python (>=2.7), python-twisted, python-w3lib, python-lxml, python-six (>=1.5.2) -Standards-Version: 3.8.4 -Homepage: https://scrapy.org/ - -Package: scrapy -Architecture: all -Depends: ${python:Depends}, python-lxml, python-twisted, python-openssl, - python-w3lib (>= 1.8.0), python-queuelib, python-cssselect (>= 0.9), python-six (>=1.5.2) -Recommends: python-setuptools -Conflicts: python-scrapy, scrapy-0.25 -Provides: python-scrapy, scrapy-0.25 -Description: Python web crawling and web scraping framework - Scrapy is a fast high-level web crawling and web scraping framework, - used to crawl websites and extract structured data from their pages. - It can be used for a wide range of purposes, from data mining to - monitoring and automated testing. diff --git a/debian/copyright b/debian/copyright deleted file mode 100644 index c1bf47565..000000000 --- a/debian/copyright +++ /dev/null @@ -1,40 +0,0 @@ -This package was debianized by the Scrapinghub team <info@scrapinghub.com>. - -It was downloaded from https://scrapy.org - -Upstream Author: Scrapy Developers - -Copyright: 2007-2013 Scrapy Developers - -License: bsd - -Copyright (c) Scrapy developers. -All rights reserved. - -Redistribution and use in source and binary forms, with or without modification, -are permitted provided that the following conditions are met: - - 1. Redistributions of source code must retain the above copyright notice, - this list of conditions and the following disclaimer. - - 2. Redistributions in binary form must reproduce the above copyright - notice, this list of conditions and the following disclaimer in the - documentation and/or other materials provided with the distribution. - - 3. Neither the name of Scrapy nor the names of its contributors may be used - to endorse or promote products derived from this software without - specific prior written permission. - -THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND CONTRIBUTORS "AS IS" AND -ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED -WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE ARE -DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR CONTRIBUTORS BE LIABLE FOR -ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL DAMAGES -(INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; -LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON -ANY THEORY OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT -(INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE OF THIS -SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE. - -The Debian packaging is (C) 2010-2013, Scrapinghub <info@scrapinghub.com> and -is licensed under the BSD, see `/usr/share/common-licenses/BSD'. diff --git a/debian/pyversions b/debian/pyversions deleted file mode 100644 index 1effb0034..000000000 --- a/debian/pyversions +++ /dev/null @@ -1 +0,0 @@ -2.7 diff --git a/debian/rules b/debian/rules deleted file mode 100755 index b8796e6e3..000000000 --- a/debian/rules +++ /dev/null @@ -1,5 +0,0 @@ -#!/usr/bin/make -f -# -*- makefile -*- - -%: - dh $@ diff --git a/debian/scrapy.docs b/debian/scrapy.docs deleted file mode 100644 index c19ffba4d..000000000 --- a/debian/scrapy.docs +++ /dev/null @@ -1,2 +0,0 @@ -README.rst -AUTHORS diff --git a/debian/scrapy.install b/debian/scrapy.install deleted file mode 100644 index c288ebed3..000000000 --- a/debian/scrapy.install +++ /dev/null @@ -1,2 +0,0 @@ -extras/scrapy_bash_completion etc/bash_completion.d/ -extras/scrapy_zsh_completion /usr/share/zsh/vendor-completions/_scrapy diff --git a/debian/scrapy.lintian-overrides b/debian/scrapy.lintian-overrides deleted file mode 100644 index b5de7f67d..000000000 --- a/debian/scrapy.lintian-overrides +++ /dev/null @@ -1 +0,0 @@ -new-package-should-close-itp-bug diff --git a/debian/scrapy.manpages b/debian/scrapy.manpages deleted file mode 100644 index 4818e9c92..000000000 --- a/debian/scrapy.manpages +++ /dev/null @@ -1 +0,0 @@ -extras/scrapy.1 From 6c35baae25517ed942c72d14d93363a389e5a9d3 Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Wed, 4 Mar 2020 00:40:11 +0000 Subject: [PATCH 246/313] Remove deprecated SiteNode and MultiValueDict classes --- scrapy/utils/datatypes.py | 174 -------------------------------------- 1 file changed, 174 deletions(-) diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index b07f995cf..175f92d77 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -6,183 +6,9 @@ This module must not depend on any module outside the Standard Library. """ import collections -import copy -import warnings import weakref from collections.abc import Mapping -from scrapy.exceptions import ScrapyDeprecationWarning - - -class MultiValueDictKeyError(KeyError): - def __init__(self, *args, **kwargs): - warnings.warn( - "scrapy.utils.datatypes.MultiValueDictKeyError is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2 - ) - super(MultiValueDictKeyError, self).__init__(*args, **kwargs) - - -class MultiValueDict(dict): - """ - A subclass of dictionary customized to handle multiple values for the same key. - - >>> d = MultiValueDict({'name': ['Adrian', 'Simon'], 'position': ['Developer']}) - >>> d['name'] - 'Simon' - >>> d.getlist('name') - ['Adrian', 'Simon'] - >>> d.get('lastname', 'nonexistent') - 'nonexistent' - >>> d.setlist('lastname', ['Holovaty', 'Willison']) - - This class exists to solve the irritating problem raised by cgi.parse_qs, - which returns a list for every key, even though most Web forms submit - single name-value pairs. - """ - def __init__(self, key_to_list_mapping=()): - warnings.warn("scrapy.utils.datatypes.MultiValueDict is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2) - dict.__init__(self, key_to_list_mapping) - - def __repr__(self): - return "<%s: %s>" % (self.__class__.__name__, dict.__repr__(self)) - - def __getitem__(self, key): - """ - Returns the last data value for this key, or [] if it's an empty list; - raises KeyError if not found. - """ - try: - list_ = dict.__getitem__(self, key) - except KeyError: - raise MultiValueDictKeyError("Key %r not found in %r" % (key, self)) - try: - return list_[-1] - except IndexError: - return [] - - def __setitem__(self, key, value): - dict.__setitem__(self, key, [value]) - - def __copy__(self): - return self.__class__(dict.items(self)) - - def __deepcopy__(self, memo=None): - if memo is None: - memo = {} - result = self.__class__() - memo[id(self)] = result - for key, value in dict.items(self): - dict.__setitem__(result, copy.deepcopy(key, memo), copy.deepcopy(value, memo)) - return result - - def get(self, key, default=None): - "Returns the default value if the requested data doesn't exist" - try: - val = self[key] - except KeyError: - return default - if val == []: - return default - return val - - def getlist(self, key): - "Returns an empty list if the requested data doesn't exist" - try: - return dict.__getitem__(self, key) - except KeyError: - return [] - - def setlist(self, key, list_): - dict.__setitem__(self, key, list_) - - def setdefault(self, key, default=None): - if key not in self: - self[key] = default - return self[key] - - def setlistdefault(self, key, default_list=()): - if key not in self: - self.setlist(key, default_list) - return self.getlist(key) - - def appendlist(self, key, value): - "Appends an item to the internal list associated with key" - self.setlistdefault(key, []) - dict.__setitem__(self, key, self.getlist(key) + [value]) - - def items(self): - """ - Returns a list of (key, value) pairs, where value is the last item in - the list associated with the key. - """ - return [(key, self[key]) for key in self.keys()] - - def lists(self): - "Returns a list of (key, list) pairs." - return dict.items(self) - - def values(self): - "Returns a list of the last value on every key list." - return [self[key] for key in self.keys()] - - def copy(self): - "Returns a copy of this object." - return self.__deepcopy__() - - def update(self, *args, **kwargs): - "update() extends rather than replaces existing key lists. Also accepts keyword args." - if len(args) > 1: - raise TypeError("update expected at most 1 arguments, got %d" % len(args)) - if args: - other_dict = args[0] - if isinstance(other_dict, MultiValueDict): - for key, value_list in other_dict.lists(): - self.setlistdefault(key, []).extend(value_list) - else: - try: - for key, value in other_dict.items(): - self.setlistdefault(key, []).append(value) - except TypeError: - raise ValueError("MultiValueDict.update() takes either a MultiValueDict or dictionary") - for key, value in kwargs.items(): - self.setlistdefault(key, []).append(value) - - -class SiteNode(object): - """Class to represent a site node (page, image or any other file)""" - - def __init__(self, url): - warnings.warn( - "scrapy.utils.datatypes.SiteNode is deprecated " - "and will be removed in future releases.", - category=ScrapyDeprecationWarning, - stacklevel=2 - ) - - self.url = url - self.itemnames = [] - self.children = [] - self.parent = None - - def add_child(self, node): - self.children.append(node) - node.parent = self - - def to_string(self, level=0): - s = "%s%s\n" % (' ' * level, self.url) - if self.itemnames: - for n in self.itemnames: - s += "%sScraped: %s\n" % (' ' * (level + 1), n) - for node in self.children: - s += node.to_string(level + 1) - return s - class CaselessDict(dict): From b1566a696217244d2d506ce8962418943d0b1edc Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Mon, 2 Mar 2020 23:01:26 +0000 Subject: [PATCH 247/313] Remove deprecated Crawler.spiders property Deprecated since 419026615 (2014, Scrapy 0.25) --- scrapy/crawler.py | 45 ++++++++++++++++--------------------------- tests/test_crawler.py | 15 --------------- 2 files changed, 17 insertions(+), 43 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 49b8e4511..77a13d0c1 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -68,17 +68,6 @@ class Crawler: self.spider = None self.engine = None - @property - def spiders(self): - if not hasattr(self, '_spiders'): - warnings.warn("Crawler.spiders is deprecated, use " - "CrawlerRunner.spider_loader or instantiate " - "scrapy.spiderloader.SpiderLoader with your " - "settings.", - category=ScrapyDeprecationWarning, stacklevel=2) - self._spiders = _get_spider_loader(self.settings.frozencopy()) - return self._spiders - @defer.inlineCallbacks def crawl(self, *args, **kwargs): assert not self.crawling, "Crawling already taking place" @@ -130,11 +119,27 @@ class CrawlerRunner: ":meth:`crawl` and managed by this class." ) + @staticmethod + def _get_spider_loader(settings): + """ Get SpiderLoader instance from settings """ + cls_path = settings.get('SPIDER_LOADER_CLASS') + loader_cls = load_object(cls_path) + try: + verifyClass(ISpiderLoader, loader_cls) + except DoesNotImplement: + warnings.warn( + 'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does ' + 'not fully implement scrapy.interfaces.ISpiderLoader interface. ' + 'Please add all missing methods to avoid unexpected runtime errors.', + category=ScrapyDeprecationWarning, stacklevel=2 + ) + return loader_cls.from_settings(settings.frozencopy()) + def __init__(self, settings=None): if isinstance(settings, dict) or settings is None: settings = Settings(settings) self.settings = settings - self.spider_loader = _get_spider_loader(settings) + self.spider_loader = self._get_spider_loader(settings) self._crawlers = set() self._active = set() self.bootstrap_failed = False @@ -327,19 +332,3 @@ class CrawlerProcess(CrawlerRunner): if self.settings.get("TWISTED_REACTOR"): install_reactor(self.settings["TWISTED_REACTOR"]) super()._handle_twisted_reactor() - - -def _get_spider_loader(settings): - """ Get SpiderLoader instance from settings """ - cls_path = settings.get('SPIDER_LOADER_CLASS') - loader_cls = load_object(cls_path) - try: - verifyClass(ISpiderLoader, loader_cls) - except DoesNotImplement: - warnings.warn( - 'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does ' - 'not fully implement scrapy.interfaces.ISpiderLoader interface. ' - 'Please add all missing methods to avoid unexpected runtime errors.', - category=ScrapyDeprecationWarning, stacklevel=2 - ) - return loader_cls.from_settings(settings.frozencopy()) diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 7bd76601d..37a069611 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -33,21 +33,6 @@ class CrawlerTestCase(BaseCrawlerTest): def setUp(self): self.crawler = Crawler(DefaultSpider, Settings()) - def test_deprecated_attribute_spiders(self): - with warnings.catch_warnings(record=True) as w: - spiders = self.crawler.spiders - self.assertEqual(len(w), 1) - self.assertIn("Crawler.spiders", str(w[0].message)) - sl_cls = load_object(self.crawler.settings['SPIDER_LOADER_CLASS']) - self.assertIsInstance(spiders, sl_cls) - - self.crawler.spiders - is_one_warning = len(w) == 1 - if not is_one_warning: - for warning in w: - print(warning) - self.assertTrue(is_one_warning, "Warn deprecated access only once") - def test_populate_spidercls_settings(self): spider_settings = {'TEST1': 'spider', 'TEST2': 'spider'} project_settings = {'TEST1': 'project', 'TEST3': 'project'} From ada37c5409047291ee5852fb2220e8e256424402 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sat, 6 Jul 2019 22:37:31 -0300 Subject: [PATCH 248/313] Export to multiple formats in a single crawl --- docs/topics/feed-exports.rst | 85 ++++++--- scrapy/commands/crawl.py | 23 +-- scrapy/commands/runspider.py | 20 +- scrapy/extensions/feedexport.py | 191 +++++++++++-------- scrapy/settings/default_settings.py | 3 +- scrapy/utils/conf.py | 68 ++++++- tests/test_commands.py | 32 +++- tests/test_feedexport.py | 273 ++++++++++++++++++++-------- tests/test_utils_conf.py | 50 ++++- 9 files changed, 511 insertions(+), 234 deletions(-) diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index 42f1cad90..6d6ba33c9 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -12,7 +12,7 @@ generating an "export file" with the scraped data (commonly called "export feed") to be consumed by other systems. Scrapy provides this functionality out of the box with the Feed Exports, which -allows you to generate a feed with the scraped items, using multiple +allows you to generate feeds with the scraped items, using multiple serialization formats and storage backends. .. _topics-feed-format: @@ -36,7 +36,7 @@ But you can also extend the supported format through the JSON ---- - * :setting:`FEED_FORMAT`: ``json`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``json`` * Exporter used: :class:`~scrapy.exporters.JsonItemExporter` * See :ref:`this warning <json-with-large-data>` if you're using JSON with large feeds. @@ -46,7 +46,7 @@ JSON JSON lines ---------- - * :setting:`FEED_FORMAT`: ``jsonlines`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``jsonlines`` * Exporter used: :class:`~scrapy.exporters.JsonLinesItemExporter` .. _topics-feed-format-csv: @@ -54,7 +54,7 @@ JSON lines CSV --- - * :setting:`FEED_FORMAT`: ``csv`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``csv`` * Exporter used: :class:`~scrapy.exporters.CsvItemExporter` * To specify columns to export and their order use :setting:`FEED_EXPORT_FIELDS`. Other feed exporters can also use this @@ -66,7 +66,7 @@ CSV XML --- - * :setting:`FEED_FORMAT`: ``xml`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``xml`` * Exporter used: :class:`~scrapy.exporters.XmlItemExporter` .. _topics-feed-format-pickle: @@ -74,7 +74,7 @@ XML Pickle ------ - * :setting:`FEED_FORMAT`: ``pickle`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``pickle`` * Exporter used: :class:`~scrapy.exporters.PickleItemExporter` .. _topics-feed-format-marshal: @@ -82,7 +82,7 @@ Pickle Marshal ------- - * :setting:`FEED_FORMAT`: ``marshal`` + * Value for the ``format`` key in the :setting:`FEEDS` setting: ``marshal`` * Exporter used: :class:`~scrapy.exporters.MarshalItemExporter` @@ -91,8 +91,8 @@ Marshal Storages ======== -When using the feed exports you define where to store the feed using a URI_ -(through the :setting:`FEED_URI` setting). The feed exports supports multiple +When using the feed exports you define where to store the feed using one or multiple URIs_ +(through the :setting:`FEEDS` setting). The feed exports supports multiple storage backend types which are defined by the URI scheme. The storages backends supported out of the box are: @@ -211,41 +211,66 @@ Settings These are the settings used for configuring the feed exports: - * :setting:`FEED_URI` (mandatory) - * :setting:`FEED_FORMAT` + * :setting:`FEEDS` (mandatory) + * :setting:`FEED_EXPORT_ENCODING` + * :setting:`FEED_STORE_EMPTY` + * :setting:`FEED_EXPORT_FIELDS` + * :setting:`FEED_EXPORT_INDENT` * :setting:`FEED_STORAGES` * :setting:`FEED_STORAGE_FTP_ACTIVE` * :setting:`FEED_STORAGE_S3_ACL` * :setting:`FEED_EXPORTERS` - * :setting:`FEED_STORE_EMPTY` - * :setting:`FEED_EXPORT_ENCODING` - * :setting:`FEED_EXPORT_FIELDS` - * :setting:`FEED_EXPORT_INDENT` .. currentmodule:: scrapy.extensions.feedexport -.. setting:: FEED_URI +.. setting:: FEEDS -FEED_URI --------- +FEEDS +----- -Default: ``None`` +.. versionadded:: 2.1 -The URI of the export feed. See :ref:`topics-feed-storage-backends` for -supported URI schemes. +Default: ``{}`` -This setting is required for enabling the feed exports. +A dictionary in which every key is a feed URI (or a :class:`pathlib.Path` +object) and each value is a nested dictionary containing configuration +parameters for the specific feed. +This setting is required for enabling the feed export feature. -.. versionchanged:: 2.0 - Added :class:`pathlib.Path` support. +See :ref:`topics-feed-storage-backends` for supported URI schemes. -.. setting:: FEED_FORMAT +For instance:: -FEED_FORMAT ------------ + { + 'items.json': { + 'format': 'json', + 'encoding': 'utf8', + 'store_empty': False, + 'fields': None, + 'indent': 4, + }, + 'items.xml': { + 'format': 'xml', + 'fields': ['name', 'price'], + 'encoding': 'latin1', + 'indent': 8, + }, + pathlib.Path('items.csv'): { + 'format': 'csv', + 'fields': ['price', 'name'], + }, + } -The serialization format to be used for the feed. See -:ref:`topics-feed-format` for possible values. +The following is a list of the accepted keys and the setting that is used +as a fallback value if that key is not provided for a specific feed definition. + +* ``format``: the serialization format to be used for the feed. + See :ref:`topics-feed-format` for possible values. + Mandatory, no fallback setting +* ``encoding``: falls back to :setting:`FEED_EXPORT_ENCODING` +* ``fields``: falls back to :setting:`FEED_EXPORT_FIELDS` +* ``indent``: falls back to :setting:`FEED_EXPORT_INDENT` +* ``store_empty``: falls back to :setting:`FEED_STORE_EMPTY` .. setting:: FEED_EXPORT_ENCODING @@ -400,7 +425,7 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter 'csv': None, } -.. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier +.. _URIs: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier .. _Amazon S3: https://aws.amazon.com/s3/ .. _botocore: https://github.com/boto/botocore .. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index 7b417e2eb..4b2f9484b 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -1,7 +1,5 @@ -import os from scrapy.commands import ScrapyCommand -from scrapy.utils.conf import arglist_to_dict -from scrapy.utils.python import without_none_values +from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli from scrapy.exceptions import UsageError @@ -19,7 +17,7 @@ class Command(ScrapyCommand): ScrapyCommand.add_options(self, parser) parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE", help="set spider argument (may be repeated)") - parser.add_option("-o", "--output", metavar="FILE", + parser.add_option("-o", "--output", metavar="FILE", action="append", help="dump scraped items into FILE (use - for stdout)") parser.add_option("-t", "--output-format", metavar="FORMAT", help="format to use for dumping items with -o") @@ -31,21 +29,8 @@ class Command(ScrapyCommand): except ValueError: raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False) if opts.output: - if opts.output == '-': - self.settings.set('FEED_URI', 'stdout:', priority='cmdline') - else: - self.settings.set('FEED_URI', opts.output, priority='cmdline') - feed_exporters = without_none_values( - self.settings.getwithbase('FEED_EXPORTERS')) - valid_output_formats = feed_exporters.keys() - if not opts.output_format: - opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") - if opts.output_format not in valid_output_formats: - raise UsageError("Unrecognized output format '%s', set one" - " using the '-t' switch or as a file extension" - " from the supported list %s" % (opts.output_format, - tuple(valid_output_formats))) - self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline') + feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format) + self.settings.set('FEEDS', feeds, priority='cmdline') def run(self, args, opts): if len(args) < 1: diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index 57d8471ca..62510609a 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -5,8 +5,7 @@ from importlib import import_module from scrapy.utils.spider import iter_spider_classes from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError -from scrapy.utils.conf import arglist_to_dict -from scrapy.utils.python import without_none_values +from scrapy.utils.conf import arglist_to_dict, feed_process_params_from_cli def _import_file(filepath): @@ -43,7 +42,7 @@ class Command(ScrapyCommand): ScrapyCommand.add_options(self, parser) parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE", help="set spider argument (may be repeated)") - parser.add_option("-o", "--output", metavar="FILE", + parser.add_option("-o", "--output", metavar="FILE", action="append", help="dump scraped items into FILE (use - for stdout)") parser.add_option("-t", "--output-format", metavar="FORMAT", help="format to use for dumping items with -o") @@ -55,19 +54,8 @@ class Command(ScrapyCommand): except ValueError: raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False) if opts.output: - if opts.output == '-': - self.settings.set('FEED_URI', 'stdout:', priority='cmdline') - else: - self.settings.set('FEED_URI', opts.output, priority='cmdline') - feed_exporters = without_none_values(self.settings.getwithbase('FEED_EXPORTERS')) - if not opts.output_format: - opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") - if opts.output_format not in feed_exporters: - raise UsageError("Unrecognized output format '%s', set one" - " using the '-t' switch or as a file extension" - " from the supported list %s" % (opts.output_format, - tuple(feed_exporters))) - self.settings.set('FEED_FORMAT', opts.output_format, priority='cmdline') + feeds = feed_process_params_from_cli(self.settings, opts.output, opts.output_format) + self.settings.set('FEEDS', feeds, priority='cmdline') def run(self, args, opts): if len(args) != 1: diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index f1b101780..108b6d35c 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -4,24 +4,27 @@ Feed Exports extension See documentation in docs/topics/feed-exports.rst """ +import logging import os import sys -import logging -from tempfile import NamedTemporaryFile +import warnings from datetime import datetime -from urllib.parse import urlparse, unquote +from tempfile import NamedTemporaryFile +from urllib.parse import unquote, urlparse -from zope.interface import Interface, implementer from twisted.internet import defer, threads from w3lib.url import file_uri_to_path +from zope.interface import implementer, Interface from scrapy import signals -from scrapy.utils.ftp import ftp_store_file -from scrapy.exceptions import NotConfigured -from scrapy.utils.misc import create_instance, load_object -from scrapy.utils.log import failure_to_exc_info -from scrapy.utils.python import without_none_values +from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.utils.boto import is_botocore +from scrapy.utils.conf import feed_complete_default_values_from_settings +from scrapy.utils.ftp import ftp_store_file +from scrapy.utils.log import failure_to_exc_info +from scrapy.utils.misc import create_instance, load_object +from scrapy.utils.python import without_none_values + logger = logging.getLogger(__name__) @@ -98,8 +101,6 @@ class S3FeedStorage(BlockingFeedStorage): from scrapy.utils.project import get_project_settings settings = get_project_settings() if 'AWS_ACCESS_KEY_ID' in settings or 'AWS_SECRET_ACCESS_KEY' in settings: - import warnings - from scrapy.exceptions import ScrapyDeprecationWarning warnings.warn( "Initialising `scrapy.extensions.feedexport.S3FeedStorage` " "without AWS keys is deprecated. Please supply credentials or " @@ -178,88 +179,117 @@ class FTPFeedStorage(BlockingFeedStorage): ) -class SpiderSlot(object): - def __init__(self, file, exporter, storage, uri): +class _FeedSlot(object): + def __init__(self, file, exporter, storage, uri, format, store_empty): self.file = file self.exporter = exporter self.storage = storage + # feed params self.uri = uri + self.format = format + self.store_empty = store_empty + # flags self.itemcount = 0 + self._exporting = False + + def start_exporting(self): + if not self._exporting: + self.exporter.start_exporting() + self._exporting = True + + def finish_exporting(self): + if self._exporting: + self.exporter.finish_exporting() + self._exporting = False class FeedExporter(object): - def __init__(self, settings): - self.settings = settings - if not settings['FEED_URI']: - raise NotConfigured - self.urifmt = str(settings['FEED_URI']) - self.format = settings['FEED_FORMAT'].lower() - self.export_encoding = settings['FEED_EXPORT_ENCODING'] - self.storages = self._load_components('FEED_STORAGES') - self.exporters = self._load_components('FEED_EXPORTERS') - if not self._storage_supported(self.urifmt): - raise NotConfigured - if not self._exporter_supported(self.format): - raise NotConfigured - self.store_empty = settings.getbool('FEED_STORE_EMPTY') - self._exporting = False - self.export_fields = settings.getlist('FEED_EXPORT_FIELDS') or None - self.indent = None - if settings.get('FEED_EXPORT_INDENT') is not None: - self.indent = settings.getint('FEED_EXPORT_INDENT') - uripar = settings['FEED_URI_PARAMS'] - self._uripar = load_object(uripar) if uripar else lambda x, y: None - @classmethod def from_crawler(cls, crawler): - o = cls(crawler.settings) - o.crawler = crawler - crawler.signals.connect(o.open_spider, signals.spider_opened) - crawler.signals.connect(o.close_spider, signals.spider_closed) - crawler.signals.connect(o.item_scraped, signals.item_scraped) - return o + exporter = cls(crawler) + crawler.signals.connect(exporter.open_spider, signals.spider_opened) + crawler.signals.connect(exporter.close_spider, signals.spider_closed) + crawler.signals.connect(exporter.item_scraped, signals.item_scraped) + return exporter + + def __init__(self, crawler): + self.crawler = crawler + self.settings = crawler.settings + self.feeds = {} + self.slots = [] + + if not self.settings['FEEDS'] and not self.settings['FEED_URI']: + raise NotConfigured + + # Begin: Backward compatibility for FEED_URI and FEED_FORMAT settings + if self.settings['FEED_URI']: + warnings.warn( + 'The `FEED_URI` and `FEED_FORMAT` settings have been deprecated in favor of ' + 'the `FEEDS` setting. Please see the `FEEDS` setting docs for more details', + category=ScrapyDeprecationWarning, stacklevel=2, + ) + uri = str(self.settings['FEED_URI']) # handle pathlib.Path objects + feed = {'format': self.settings.get('FEED_FORMAT', 'jsonlines')} + self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings) + # End: Backward compatibility for FEED_URI and FEED_FORMAT settings + + # 'FEEDS' setting takes precedence over 'FEED_URI' + for uri, feed in self.settings.getdict('FEEDS').items(): + uri = str(uri) # handle pathlib.Path objects + self.feeds[uri] = feed_complete_default_values_from_settings(feed, self.settings) + + self.storages = self._load_components('FEED_STORAGES') + self.exporters = self._load_components('FEED_EXPORTERS') + for uri, feed in self.feeds.items(): + if not self._storage_supported(uri): + raise NotConfigured + if not self._exporter_supported(feed['format']): + raise NotConfigured def open_spider(self, spider): - uri = self.urifmt % self._get_uri_params(spider) - storage = self._get_storage(uri) - file = storage.open(spider) - exporter = self._get_exporter(file, fields_to_export=self.export_fields, - encoding=self.export_encoding, indent=self.indent) - if self.store_empty: - exporter.start_exporting() - self._exporting = True - self.slot = SpiderSlot(file, exporter, storage, uri) + for uri, feed in self.feeds.items(): + uri = uri % self._get_uri_params(spider, feed['uri_params']) + storage = self._get_storage(uri) + file = storage.open(spider) + exporter = self._get_exporter( + file=file, + format=feed['format'], + fields_to_export=feed['fields'], + encoding=feed['encoding'], + indent=feed['indent'], + ) + slot = _FeedSlot(file, exporter, storage, uri, feed['format'], feed['store_empty']) + self.slots.append(slot) + if slot.store_empty: + slot.start_exporting() def close_spider(self, spider): - slot = self.slot - if not slot.itemcount and not self.store_empty: - # We need to call slot.storage.store nonetheless to get the file - # properly closed. - return defer.maybeDeferred(slot.storage.store, slot.file) - if self._exporting: - slot.exporter.finish_exporting() - self._exporting = False - logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s" - log_args = {'format': self.format, - 'itemcount': slot.itemcount, - 'uri': slot.uri} - d = defer.maybeDeferred(slot.storage.store, slot.file) - d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args, - extra={'spider': spider})) - d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args, - exc_info=failure_to_exc_info(f), - extra={'spider': spider})) - return d + deferred_list = [] + for slot in self.slots: + if not slot.itemcount and not slot.store_empty: + # We need to call slot.storage.store nonetheless to get the file + # properly closed. + return defer.maybeDeferred(slot.storage.store, slot.file) + slot.finish_exporting() + logfmt = "%s %%(format)s feed (%%(itemcount)d items) in: %%(uri)s" + log_args = {'format': slot.format, + 'itemcount': slot.itemcount, + 'uri': slot.uri} + d = defer.maybeDeferred(slot.storage.store, slot.file) + d.addCallback(lambda _: logger.info(logfmt % "Stored", log_args, + extra={'spider': spider})) + d.addErrback(lambda f: logger.error(logfmt % "Error storing", log_args, + exc_info=failure_to_exc_info(f), + extra={'spider': spider})) + deferred_list.append(d) + return defer.DeferredList(deferred_list) if deferred_list else None def item_scraped(self, item, spider): - slot = self.slot - if not self._exporting: - slot.exporter.start_exporting() - self._exporting = True - slot.exporter.export_item(item) - slot.itemcount += 1 - return item + for slot in self.slots: + slot.start_exporting() + slot.exporter.export_item(item) + slot.itemcount += 1 def _load_components(self, setting_prefix): conf = without_none_values(self.settings.getwithbase(setting_prefix)) @@ -295,17 +325,18 @@ class FeedExporter(object): objcls, self.settings, getattr(self, 'crawler', None), *args, **kwargs) - def _get_exporter(self, *args, **kwargs): - return self._get_instance(self.exporters[self.format], *args, **kwargs) + def _get_exporter(self, file, format, *args, **kwargs): + return self._get_instance(self.exporters[format], file, *args, **kwargs) def _get_storage(self, uri): return self._get_instance(self.storages[urlparse(uri).scheme], uri) - def _get_uri_params(self, spider): + def _get_uri_params(self, spider, uri_params): params = {} for k in dir(spider): params[k] = getattr(spider, k) ts = datetime.utcnow().replace(microsecond=0).isoformat().replace(':', '-') params['time'] = ts - self._uripar(params, spider) + uripar_function = load_object(uri_params) if uri_params else lambda x, y: None + uripar_function(params, spider) return params diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index f8a0457ce..077317c81 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -133,9 +133,8 @@ EXTENSIONS_BASE = { } FEED_TEMPDIR = None -FEED_URI = None +FEEDS = {} FEED_URI_PARAMS = None # a function to extend uri arguments -FEED_FORMAT = 'jsonlines' FEED_STORE_EMPTY = False FEED_EXPORT_ENCODING = None FEED_EXPORT_FIELDS = None diff --git a/scrapy/utils/conf.py b/scrapy/utils/conf.py index 23306ca28..e01027491 100644 --- a/scrapy/utils/conf.py +++ b/scrapy/utils/conf.py @@ -1,9 +1,12 @@ +import numbers import os import sys -import numbers +import warnings from configparser import ConfigParser from operator import itemgetter +from scrapy.exceptions import ScrapyDeprecationWarning, UsageError + from scrapy.settings import BaseSettings from scrapy.utils.deprecate import update_classpath from scrapy.utils.python import without_none_values @@ -106,3 +109,66 @@ def get_sources(use_closest=True): if use_closest: sources.append(closest_scrapy_cfg()) return sources + + +def feed_complete_default_values_from_settings(feed, settings): + out = feed.copy() + if 'encoding' not in out: + out['encoding'] = settings['FEED_EXPORT_ENCODING'] + if 'fields' not in out: + out['fields'] = settings.getlist('FEED_EXPORT_FIELDS') or None + if 'indent' not in out: + out['indent'] = None if settings['FEED_EXPORT_INDENT'] is None else settings.getint('FEED_EXPORT_INDENT') + if 'store_empty' not in out: + out['store_empty'] = settings.getbool('FEED_STORE_EMPTY') + if 'uri_params' not in out: + out['uri_params'] = settings['FEED_URI_PARAMS'] + return out + + +def feed_process_params_from_cli(settings, output, output_format=None): + """ + Receives feed export params (from the 'crawl' or 'runspider' commands), + checks for inconsistencies in their quantities and returns a dictionary + suitable to be used as the FEEDS setting. + """ + valid_output_formats = without_none_values( + settings.getwithbase('FEED_EXPORTERS') + ).keys() + + def check_valid_format(output_format): + if output_format not in valid_output_formats: + raise UsageError("Unrecognized output format '%s', set one after a" + " colon using the -o option (i.e. -o <URI>:<FORMAT>)" + " or as a file extension, from the supported list %s" % + (output_format, tuple(valid_output_formats))) + + if output_format: + if len(output) == 1: + check_valid_format(output_format) + warnings.warn('The -t command line option is deprecated in favor' + ' of specifying the output format within the -o' + ' option, please check the -o option docs for more details', + category=ScrapyDeprecationWarning, stacklevel=2) + return {output[0]: {'format': output_format}} + else: + raise UsageError('The -t command line option cannot be used if multiple' + ' output files are specified with the -o option') + + result = {} + for element in output: + try: + feed_uri, feed_format = element.rsplit(':', 1) + except ValueError: + feed_uri = element + feed_format = os.path.splitext(element)[1].replace('.', '') + else: + if feed_uri == '-': + feed_uri = 'stdout:' + check_valid_format(feed_format) + result[feed_uri] = {'format': feed_format} + + # FEEDS setting should take precedence over the -o and -t CLI options + result.update(settings.getdict('FEEDS')) + + return result diff --git a/tests/test_commands.py b/tests/test_commands.py index 3612b70c9..24a341759 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -1,22 +1,46 @@ import inspect +import json +import optparse import os -import sys import subprocess +import sys import tempfile +from contextlib import contextmanager from os.path import exists, join, abspath from shutil import rmtree, copytree from tempfile import mkdtemp -from contextlib import contextmanager from threading import Timer from twisted.trial import unittest import scrapy +from scrapy.commands import ScrapyCommand +from scrapy.settings import Settings from scrapy.utils.python import to_unicode from scrapy.utils.test import get_testenv + from tests.test_crawler import ExceptionSpider, NoRequestsSpider +class CommandSettings(unittest.TestCase): + + def setUp(self): + self.command = ScrapyCommand() + self.command.settings = Settings() + self.parser = optparse.OptionParser( + formatter=optparse.TitledHelpFormatter(), + conflict_handler='resolve', + ) + self.command.add_options(self.parser) + + def test_settings_json_string(self): + feeds_json = '{"data.json": {"format": "json"}, "data.xml": {"format": "xml"}}' + opts, args = self.parser.parse_args(args=['-s', 'FEEDS={}'.format(feeds_json), 'spider.py']) + self.command.process_options(args, opts) + self.assertIsInstance(self.command.settings['FEEDS'], scrapy.settings.BaseSettings) + self.assertEqual(dict(self.command.settings['FEEDS']), json.loads(feeds_json)) + + class ProjectTest(unittest.TestCase): project_name = 'testproject' @@ -34,7 +58,7 @@ class ProjectTest(unittest.TestCase): with tempfile.TemporaryFile() as out: args = (sys.executable, '-m', 'scrapy.cmdline') + new_args return subprocess.call(args, stdout=out, stderr=out, cwd=self.cwd, - env=self.env, **kwargs) + env=self.env, **kwargs) def proc(self, *new_args, **popen_kwargs): args = (sys.executable, '-m', 'scrapy.cmdline') + new_args @@ -310,6 +334,6 @@ class BenchCommandTest(CommandTest): def test_run(self): _, _, log = self.proc('bench', '-s', 'LOGSTATS_INTERVAL=0.001', - '-s', 'CLOSESPIDER_TIMEOUT=0.01') + '-s', 'CLOSESPIDER_TIMEOUT=0.01') self.assertIn('INFO: Crawled', log) self.assertNotIn('Unhandled Error', log) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 2ca57c19d..08e8dfc41 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -1,32 +1,34 @@ -import os import csv import json -import warnings -import tempfile +import os +import random import shutil import string +import tempfile +import warnings from io import BytesIO from pathlib import Path +from string import ascii_letters, digits from unittest import mock from urllib.parse import urljoin, urlparse, quote from urllib.request import pathname2url -from zope.interface.verify import verifyObject -from twisted.trial import unittest +import lxml.etree from twisted.internet import defer -from scrapy.crawler import CrawlerRunner -from scrapy.settings import Settings -from tests.mockserver import MockServer +from twisted.trial import unittest from w3lib.url import path_to_file_uri +from zope.interface.verify import verifyObject import scrapy +from scrapy.crawler import CrawlerRunner from scrapy.exporters import CsvItemExporter -from scrapy.extensions.feedexport import ( - IFeedStorage, FileFeedStorage, FTPFeedStorage, - S3FeedStorage, StdoutFeedStorage, - BlockingFeedStorage) -from scrapy.utils.test import assert_aws_environ, get_s3_content_and_delete, get_crawler +from scrapy.extensions.feedexport import (BlockingFeedStorage, FileFeedStorage, FTPFeedStorage, + IFeedStorage, S3FeedStorage, StdoutFeedStorage) +from scrapy.settings import Settings from scrapy.utils.python import to_unicode +from scrapy.utils.test import assert_aws_environ, get_crawler, get_s3_content_and_delete + +from tests.mockserver import MockServer class FileFeedStorageTest(unittest.TestCase): @@ -395,29 +397,41 @@ class FeedExportTest(unittest.TestCase): egg = scrapy.Field() baz = scrapy.Field() + def setUp(self): + self.temp_dir = tempfile.mkdtemp() + + def tearDown(self): + shutil.rmtree(self.temp_dir, ignore_errors=True) + + def _random_temp_filename(self): + chars = [random.choice(ascii_letters + digits) for _ in range(15)] + filename = ''.join(chars) + return os.path.join(self.temp_dir, filename) + @defer.inlineCallbacks - def run_and_export(self, spider_cls, settings=None): + def run_and_export(self, spider_cls, settings): """ Run spider with specified settings; return exported data. """ - tmpdir = tempfile.mkdtemp() - res_path = os.path.join(tmpdir, 'res') - res_uri = urljoin('file:', pathname2url(res_path)) - defaults = { - 'FEED_URI': res_uri, - 'FEED_FORMAT': 'csv', - 'FEED_PATH': res_path + + FEEDS = settings.get('FEEDS') or {} + settings['FEEDS'] = { + urljoin('file:', pathname2url(str(file_path))): feed + for file_path, feed in FEEDS.items() } - defaults.update(settings or {}) + + content = {} try: with MockServer() as s: - runner = CrawlerRunner(Settings(defaults)) + runner = CrawlerRunner(Settings(settings)) spider_cls.start_urls = [s.url('/')] yield runner.crawl(spider_cls) - with open(str(defaults['FEED_PATH']), 'rb') as f: - content = f.read() + for file_path, feed in FEEDS.items(): + with open(str(file_path), 'rb') as f: + content[feed['format']] = f.read() finally: - shutil.rmtree(tmpdir) + for file_path in FEEDS.keys(): + os.remove(str(file_path)) defer.returnValue(content) @@ -453,10 +467,14 @@ class FeedExportTest(unittest.TestCase): @defer.inlineCallbacks def assertExportedCsv(self, items, header, rows, settings=None, ordered=True): settings = settings or {} - settings.update({'FEED_FORMAT': 'csv'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'csv'}, + }, + }) data = yield self.exported_data(items, settings) - reader = csv.DictReader(to_unicode(data).splitlines()) + reader = csv.DictReader(to_unicode(data['csv']).splitlines()) got_rows = list(reader) if ordered: self.assertEqual(reader.fieldnames, header) @@ -468,51 +486,87 @@ class FeedExportTest(unittest.TestCase): @defer.inlineCallbacks def assertExportedJsonLines(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'jl'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'jl'}, + }, + }) data = yield self.exported_data(items, settings) - parsed = [json.loads(to_unicode(line)) for line in data.splitlines()] + parsed = [json.loads(to_unicode(line)) for line in data['jl'].splitlines()] rows = [{k: v for k, v in row.items() if v} for row in rows] self.assertEqual(rows, parsed) @defer.inlineCallbacks def assertExportedXml(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'xml'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'xml'}, + }, + }) data = yield self.exported_data(items, settings) rows = [{k: v for k, v in row.items() if v} for row in rows] - import lxml.etree - root = lxml.etree.fromstring(data) + root = lxml.etree.fromstring(data['xml']) got_rows = [{e.tag: e.text for e in it} for it in root.findall('item')] self.assertEqual(rows, got_rows) + @defer.inlineCallbacks + def assertExportedMultiple(self, items, rows, settings=None): + settings = settings or {} + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'xml'}, + self._random_temp_filename(): {'format': 'json'}, + }, + }) + data = yield self.exported_data(items, settings) + rows = [{k: v for k, v in row.items() if v} for row in rows] + # XML + root = lxml.etree.fromstring(data['xml']) + xml_rows = [{e.tag: e.text for e in it} for it in root.findall('item')] + self.assertEqual(rows, xml_rows) + # JSON + json_rows = json.loads(to_unicode(data['json'])) + self.assertEqual(rows, json_rows) + def _load_until_eof(self, data, load_func): - bytes_output = BytesIO(data) result = [] - while True: - try: - result.append(load_func(bytes_output)) - except EOFError: - break + with tempfile.TemporaryFile() as temp: + temp.write(data) + temp.seek(0) + while True: + try: + result.append(load_func(temp)) + except EOFError: + break return result @defer.inlineCallbacks def assertExportedPickle(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'pickle'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'pickle'}, + }, + }) data = yield self.exported_data(items, settings) expected = [{k: v for k, v in row.items() if v} for row in rows] import pickle - result = self._load_until_eof(data, load_func=pickle.load) + result = self._load_until_eof(data['pickle'], load_func=pickle.load) self.assertEqual(expected, result) @defer.inlineCallbacks def assertExportedMarshal(self, items, rows, settings=None): settings = settings or {} - settings.update({'FEED_FORMAT': 'marshal'}) + settings.update({ + 'FEEDS': { + self._random_temp_filename(): {'format': 'marshal'}, + }, + }) data = yield self.exported_data(items, settings) expected = [{k: v for k, v in row.items() if v} for row in rows] import marshal - result = self._load_until_eof(data, load_func=marshal.load) + result = self._load_until_eof(data['marshal'], load_func=marshal.load) self.assertEqual(expected, result) @defer.inlineCallbacks @@ -521,6 +575,8 @@ class FeedExportTest(unittest.TestCase): yield self.assertExportedJsonLines(items, rows, settings) yield self.assertExportedXml(items, rows, settings) yield self.assertExportedPickle(items, rows, settings) + yield self.assertExportedMarshal(items, rows, settings) + yield self.assertExportedMultiple(items, rows, settings) @defer.inlineCallbacks def test_export_items(self): @@ -538,15 +594,14 @@ class FeedExportTest(unittest.TestCase): @defer.inlineCallbacks def test_export_no_items_not_store_empty(self): - formats = ('json', - 'jsonlines', - 'xml', - 'csv',) - - for fmt in formats: - settings = {'FEED_FORMAT': fmt} + for fmt in ('json', 'jsonlines', 'xml', 'csv'): + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + } data = yield self.exported_no_data(settings) - self.assertEqual(data, b'') + self.assertEqual(data[fmt], b'') @defer.inlineCallbacks def test_export_no_items_store_empty(self): @@ -558,9 +613,15 @@ class FeedExportTest(unittest.TestCase): ) for fmt, expctd in formats: - settings = {'FEED_FORMAT': fmt, 'FEED_STORE_EMPTY': True, 'FEED_EXPORT_INDENT': None} + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + 'FEED_STORE_EMPTY': True, + 'FEED_EXPORT_INDENT': None, + } data = yield self.exported_no_data(settings) - self.assertEqual(data, expctd) + self.assertEqual(data[fmt], expctd) @defer.inlineCallbacks def test_export_multiple_item_classes(self): @@ -581,9 +642,9 @@ class FeedExportTest(unittest.TestCase): header = self.MyItem.fields.keys() rows_csv = [ {'egg': 'spam1', 'foo': 'bar1', 'baz': ''}, - {'egg': '', 'foo': 'bar2', 'baz': ''}, + {'egg': '', 'foo': 'bar2', 'baz': ''}, {'egg': 'spam3', 'foo': 'bar3', 'baz': 'quux3'}, - {'egg': 'spam4', 'foo': '', 'baz': ''}, + {'egg': 'spam4', 'foo': '', 'baz': ''}, ] rows_jl = [dict(row) for row in items] yield self.assertExportedCsv(items, header, rows_csv, ordered=False) @@ -598,10 +659,10 @@ class FeedExportTest(unittest.TestCase): header = ["foo", "baz", "hello"] settings = {'FEED_EXPORT_FIELDS': header} rows = [ - {'foo': 'bar1', 'baz': '', 'hello': ''}, - {'foo': 'bar2', 'baz': '', 'hello': 'world2'}, + {'foo': 'bar1', 'baz': '', 'hello': ''}, + {'foo': 'bar2', 'baz': '', 'hello': 'world2'}, {'foo': 'bar3', 'baz': 'quux3', 'hello': ''}, - {'foo': '', 'baz': '', 'hello': 'world4'}, + {'foo': '', 'baz': '', 'hello': 'world4'}, ] yield self.assertExported(items, header, rows, settings=settings, ordered=True) @@ -663,10 +724,15 @@ class FeedExportTest(unittest.TestCase): 'csv': u'foo\r\nTest\xd6\r\n'.encode('utf-8'), } - for format, expected in formats.items(): - settings = {'FEED_FORMAT': format, 'FEED_EXPORT_INDENT': None} + for fmt, expected in formats.items(): + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + 'FEED_EXPORT_INDENT': None, + } data = yield self.exported_data(items, settings) - self.assertEqual(expected, data) + self.assertEqual(expected, data[fmt]) formats = { 'json': u'[{"foo": "Test\xd6"}]'.encode('latin-1'), @@ -675,11 +741,53 @@ class FeedExportTest(unittest.TestCase): 'csv': u'foo\r\nTest\xd6\r\n'.encode('latin-1'), } - settings = {'FEED_EXPORT_INDENT': None, 'FEED_EXPORT_ENCODING': 'latin-1'} - for format, expected in formats.items(): - settings['FEED_FORMAT'] = format + for fmt, expected in formats.items(): + settings = { + 'FEEDS': { + self._random_temp_filename(): {'format': fmt}, + }, + 'FEED_EXPORT_INDENT': None, + 'FEED_EXPORT_ENCODING': 'latin-1', + } data = yield self.exported_data(items, settings) - self.assertEqual(expected, data) + self.assertEqual(expected, data[fmt]) + + @defer.inlineCallbacks + def test_export_multiple_configs(self): + items = [dict({'foo': u'FOO', 'bar': u'BAR'})] + + formats = { + 'json': u'[\n{"bar": "BAR"}\n]'.encode('utf-8'), + 'xml': u'<?xml version="1.0" encoding="latin-1"?>\n<items>\n <item>\n <foo>FOO</foo>\n </item>\n</items>'.encode('latin-1'), + 'csv': u'bar,foo\r\nBAR,FOO\r\n'.encode('utf-8'), + } + + settings = { + 'FEEDS': { + self._random_temp_filename(): { + 'format': 'json', + 'indent': 0, + 'fields': ['bar'], + 'encoding': 'utf-8', + }, + self._random_temp_filename(): { + 'format': 'xml', + 'indent': 2, + 'fields': ['foo'], + 'encoding': 'latin-1', + }, + self._random_temp_filename(): { + 'format': 'csv', + 'indent': None, + 'fields': ['bar', 'foo'], + 'encoding': 'utf-8', + }, + }, + } + + data = yield self.exported_data(items, settings) + for fmt, expected in formats.items(): + self.assertEqual(expected, data[fmt]) @defer.inlineCallbacks def test_export_indentation(self): @@ -827,33 +935,38 @@ class FeedExportTest(unittest.TestCase): ] for row in test_cases: - settings = {'FEED_FORMAT': row['format'], 'FEED_EXPORT_INDENT': row['indent']} + settings = { + 'FEEDS': { + self._random_temp_filename(): { + 'format': row['format'], + 'indent': row['indent'], + }, + }, + } data = yield self.exported_data(items, settings) - print(row['format'], row['indent']) - self.assertEqual(row['expected'], data) + self.assertEqual(row['expected'], data[row['format']]) @defer.inlineCallbacks def test_init_exporters_storages_with_crawler(self): settings = { - 'FEED_EXPORTERS': {'csv': 'tests.test_feedexport.' - 'FromCrawlerCsvItemExporter'}, - 'FEED_STORAGES': {'file': 'tests.test_feedexport.' - 'FromCrawlerFileFeedStorage'}, + 'FEED_EXPORTERS': {'csv': 'tests.test_feedexport.FromCrawlerCsvItemExporter'}, + 'FEED_STORAGES': {'file': 'tests.test_feedexport.FromCrawlerFileFeedStorage'}, + 'FEEDS': { + self._random_temp_filename(): {'format': 'csv'}, + }, } - yield self.exported_data({}, settings) + yield self.exported_data(items=[], settings=settings) self.assertTrue(FromCrawlerCsvItemExporter.init_with_crawler) self.assertTrue(FromCrawlerFileFeedStorage.init_with_crawler) @defer.inlineCallbacks def test_pathlib_uri(self): - tmpdir = tempfile.mkdtemp() - feed_uri = Path(tmpdir) / 'res' + feed_path = Path(self._random_temp_filename()) settings = { - 'FEED_FORMAT': 'csv', 'FEED_STORE_EMPTY': True, - 'FEED_URI': feed_uri, - 'FEED_PATH': feed_uri + 'FEEDS': { + feed_path: {'format': 'csv'} + }, } data = yield self.exported_no_data(settings) - self.assertEqual(data, b'') - shutil.rmtree(tmpdir, ignore_errors=True) + self.assertEqual(data['csv'], b'') diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index 61e110845..f064a646c 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -1,7 +1,9 @@ import unittest +import warnings -from scrapy.settings import BaseSettings -from scrapy.utils.conf import build_component_list, arglist_to_dict +from scrapy.exceptions import UsageError, ScrapyDeprecationWarning +from scrapy.settings import BaseSettings, Settings +from scrapy.utils.conf import build_component_list, arglist_to_dict, feed_process_params_from_cli class BuildComponentListTest(unittest.TestCase): @@ -90,5 +92,49 @@ class UtilsConfTestCase(unittest.TestCase): {'arg1': 'val1', 'arg2': 'val2'}) +class FeedExportConfigTestCase(unittest.TestCase): + + def test_feed_export_config_invalid_format(self): + settings = Settings() + self.assertRaises(UsageError, feed_process_params_from_cli, settings, ['items.dat'], 'noformat') + + def test_feed_export_config_mismatch(self): + settings = Settings() + self.assertRaises( + UsageError, + feed_process_params_from_cli, settings, ['items1.dat', 'items2.dat'], 'noformat' + ) + + def test_feed_export_config_backward_compatible(self): + with warnings.catch_warnings(record=True) as cw: + settings = Settings() + self.assertEqual( + {'items.dat': {'format': 'csv'}}, + feed_process_params_from_cli(settings, ['items.dat'], 'csv') + ) + self.assertEqual(cw[0].category, ScrapyDeprecationWarning) + + def test_feed_export_config_explicit_formats(self): + settings = Settings() + self.assertEqual( + {'items_1.dat': {'format': 'json'}, 'items_2.dat': {'format': 'xml'}, 'items_3.dat': {'format': 'csv'}}, + feed_process_params_from_cli(settings, ['items_1.dat:json', 'items_2.dat:xml', 'items_3.dat:csv']) + ) + + def test_feed_export_config_implicit_formats(self): + settings = Settings() + self.assertEqual( + {'items_1.json': {'format': 'json'}, 'items_2.xml': {'format': 'xml'}, 'items_3.csv': {'format': 'csv'}}, + feed_process_params_from_cli(settings, ['items_1.json', 'items_2.xml', 'items_3.csv']) + ) + + def test_feed_export_config_stdout(self): + settings = Settings() + self.assertEqual( + {'stdout:': {'format': 'pickle'}}, + feed_process_params_from_cli(settings, ['-:pickle']) + ) + + if __name__ == "__main__": unittest.main() From c2c6ea376ca2a1a0634946f95690c03fefb9990b Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Wed, 4 Mar 2020 21:30:32 +0000 Subject: [PATCH 249/313] Remove obsolete DEPRECATED_SETTINGS (deprecated.py) --- scrapy/cmdline.py | 2 -- scrapy/settings/deprecated.py | 23 ----------------------- 2 files changed, 25 deletions(-) delete mode 100644 scrapy/settings/deprecated.py diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index ec78f7c91..a4ec7c8ae 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -12,7 +12,6 @@ from scrapy.exceptions import UsageError from scrapy.utils.misc import walk_modules from scrapy.utils.project import inside_project, get_project_settings from scrapy.utils.python import garbage_collect -from scrapy.settings.deprecated import check_deprecated_settings def _iter_command_classes(module_name): @@ -118,7 +117,6 @@ def execute(argv=None, settings=None): pass else: settings['EDITOR'] = editor - check_deprecated_settings(settings) inproject = inside_project() cmds = _get_commands_dict(settings, inproject) diff --git a/scrapy/settings/deprecated.py b/scrapy/settings/deprecated.py deleted file mode 100644 index f6f878725..000000000 --- a/scrapy/settings/deprecated.py +++ /dev/null @@ -1,23 +0,0 @@ -import warnings -from scrapy.exceptions import ScrapyDeprecationWarning - -DEPRECATED_SETTINGS = [ - ('TRACK_REFS', 'no longer needed (trackref is always enabled)'), - ('RESPONSE_CLASSES', 'no longer supported'), - ('DEFAULT_RESPONSE_ENCODING', 'no longer supported'), - ('BOT_VERSION', 'no longer used (user agent defaults to Scrapy now)'), - ('ENCODING_ALIASES', 'no longer needed (encoding discovery uses w3lib now)'), - ('STATS_ENABLED', 'no longer supported (change STATS_CLASS instead)'), - ('SQLITE_DB', 'no longer supported'), - ('AUTOTHROTTLE_MIN_DOWNLOAD_DELAY', 'use DOWNLOAD_DELAY instead'), - ('AUTOTHROTTLE_MAX_CONCURRENCY', 'use CONCURRENT_REQUESTS_PER_DOMAIN instead'), -] - - -def check_deprecated_settings(settings): - deprecated = [x for x in DEPRECATED_SETTINGS if settings[x[0]] is not None] - if deprecated: - msg = "You are using the following settings which are deprecated or obsolete" - msg += " (ask scrapy-users@googlegroups.com for alternatives):" - msg = msg + "\n " + "\n ".join("%s: %s" % x for x in deprecated) - warnings.warn(msg, ScrapyDeprecationWarning) From 915e363db5cb208de9043b61015b79c91ed0a6bb Mon Sep 17 00:00:00 2001 From: nyov <nyov@nexnode.net> Date: Sat, 7 Mar 2020 18:03:25 +0000 Subject: [PATCH 250/313] Remove a 'twisted.test.proto_helpers' deprecation warning --- tests/test_webclient.py | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/tests/test_webclient.py b/tests/test_webclient.py index 99a998a46..6253d5c3f 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -9,7 +9,12 @@ import OpenSSL.SSL from twisted.trial import unittest from twisted.web import server, static, util, resource from twisted.internet import reactor, defer -from twisted.test.proto_helpers import StringTransport +try: + from twisted.internet.testing import StringTransport +except ImportError: + # deprecated in Twisted 19.7.0 + # (remove once we bump our requirement past that version) + from twisted.test.proto_helpers import StringTransport from twisted.python.filepath import FilePath from twisted.protocols.policies import WrappingFactory from twisted.internet.defer import inlineCallbacks From 9d9dea0d69709ef0f7aef67ddba1bd7bda25d273 Mon Sep 17 00:00:00 2001 From: Lukas Anzinger <lukas@lukasanzinger.at> Date: Sat, 7 Mar 2020 19:54:25 +0100 Subject: [PATCH 251/313] Fix handling of None in allowed_domains. Nones in allowed_domains ought to be ignored and there are also tests for that scenario. This commit fixes the handling of None and also the accompanying tests which are now executed again. --- scrapy/spidermiddlewares/offsite.py | 8 ++++++-- tests/test_spidermiddleware_offsite.py | 18 +++++++++--------- 2 files changed, 15 insertions(+), 11 deletions(-) diff --git a/scrapy/spidermiddlewares/offsite.py b/scrapy/spidermiddlewares/offsite.py index 232e96cbb..36f809699 100644 --- a/scrapy/spidermiddlewares/offsite.py +++ b/scrapy/spidermiddlewares/offsite.py @@ -54,12 +54,16 @@ class OffsiteMiddleware(object): if not allowed_domains: return re.compile('') # allow all by default url_pattern = re.compile("^https?://.*$") + domains = [] for domain in allowed_domains: - if url_pattern.match(domain): + if domain is None: + continue + elif url_pattern.match(domain): message = ("allowed_domains accepts only domains, not URLs. " "Ignoring URL entry %s in allowed_domains." % domain) warnings.warn(message, URLWarning) - domains = [re.escape(d) for d in allowed_domains if d is not None] + else: + domains.append(re.escape(domain)) regex = r'^(.*\.)?(%s)$' % '|'.join(domains) return re.compile(regex) diff --git a/tests/test_spidermiddleware_offsite.py b/tests/test_spidermiddleware_offsite.py index 7511aa568..51c328943 100644 --- a/tests/test_spidermiddleware_offsite.py +++ b/tests/test_spidermiddleware_offsite.py @@ -55,21 +55,21 @@ class TestOffsiteMiddleware2(TestOffsiteMiddleware): class TestOffsiteMiddleware3(TestOffsiteMiddleware2): - def _get_spider(self): - return Spider('foo') + def _get_spiderargs(self): + return dict(name='foo') class TestOffsiteMiddleware4(TestOffsiteMiddleware3): - def _get_spider(self): - bad_hostname = urlparse('http:////scrapytest.org').hostname - return dict(name='foo', allowed_domains=['scrapytest.org', None, bad_hostname]) + def _get_spiderargs(self): + bad_hostname = urlparse('http:////scrapytest.org').hostname + return dict(name='foo', allowed_domains=['scrapytest.org', None, bad_hostname]) def test_process_spider_output(self): - res = Response('http://scrapytest.org') - reqs = [Request('http://scrapytest.org/1')] - out = list(self.mw.process_spider_output(res, reqs, self.spider)) - self.assertEqual(out, reqs) + res = Response('http://scrapytest.org') + reqs = [Request('http://scrapytest.org/1')] + out = list(self.mw.process_spider_output(res, reqs, self.spider)) + self.assertEqual(out, reqs) class TestOffsiteMiddleware5(TestOffsiteMiddleware4): From 91a78eef3ee9de033e66db55c49321b2cc43740e Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sun, 8 Mar 2020 22:32:17 -0300 Subject: [PATCH 252/313] Pass callback results as dicts instead of tuples --- scrapy/core/downloader/handlers/http11.py | 56 ++++++++++++++++------- 1 file changed, 40 insertions(+), 16 deletions(-) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 190ae1d3b..e904cbc05 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -384,7 +384,13 @@ class ScrapyAgent(object): def _cb_bodyready(self, txresponse, request): # deliverBody hangs for responses without body if txresponse.length == 0: - return txresponse, b'', None, None + return { + "txresponse": txresponse, + "body": b"", + "flags": None, + "certificate": None, + "ip_address": None, + } maxsize = request.meta.get('download_maxsize', self._maxsize) warnsize = request.meta.get('download_warnsize', self._warnsize) @@ -420,12 +426,18 @@ class ScrapyAgent(object): return d def _cb_bodydone(self, result, request, url): - txresponse, body, flags, certificate, ip_address = result - status = int(txresponse.code) - headers = Headers(txresponse.headers.getAllRawHeaders()) - respcls = responsetypes.from_args(headers=headers, url=url, body=body) - return respcls(url=url, status=status, headers=headers, body=body, - flags=flags, certificate=certificate, ip_address=ip_address) + status = int(result["txresponse"].code) + headers = Headers(result["txresponse"].headers.getAllRawHeaders()) + respcls = responsetypes.from_args(headers=headers, url=url, body=result["body"]) + return respcls( + url=url, + status=status, + headers=headers, + body=result["body"], + flags=result["flags"], + certificate=result["certificate"], + ip_address=result["ip_address"], + ) @implementer(IBodyProducer) @@ -501,22 +513,34 @@ class _ResponseReader(protocol.Protocol): body = self._bodybuf.getvalue() if reason.check(ResponseDone): - self._finished.callback( - (self._txresponse, body, None, self._certificate, self._ip_address) - ) + self._finished.callback({ + "txresponse": self._txresponse, + "body": body, + "flags": None, + "certificate": self._certificate, + "ip_address": self._ip_address, + }) return if reason.check(PotentialDataLoss): - self._finished.callback( - (self._txresponse, body, ['partial'], self._certificate, self._ip_address) - ) + self._finished.callback({ + "txresponse": self._txresponse, + "body": body, + "flags": ["partial"], + "certificate": self._certificate, + "ip_address": self._ip_address, + }) return if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons): if not self._fail_on_dataloss: - self._finished.callback( - (self._txresponse, body, ['dataloss'], self._certificate, self._ip_address) - ) + self._finished.callback({ + "txresponse": self._txresponse, + "body": body, + "flags": ["dataloss"], + "certificate": self._certificate, + "ip_address": self._ip_address, + }) return elif not self._fail_on_dataloss_warned: From 1785095707dec53647c835c0b0861b220e8495af Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 11 Mar 2020 20:41:59 -0300 Subject: [PATCH 253/313] Remove single-use variable --- scrapy/core/downloader/handlers/http11.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index e904cbc05..a5b03a62b 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -426,12 +426,11 @@ class ScrapyAgent(object): return d def _cb_bodydone(self, result, request, url): - status = int(result["txresponse"].code) headers = Headers(result["txresponse"].headers.getAllRawHeaders()) respcls = responsetypes.from_args(headers=headers, url=url, body=result["body"]) return respcls( url=url, - status=status, + status=int(result["txresponse"].code), headers=headers, body=result["body"], flags=result["flags"], From 49156f2ecb0197c96e3889805a233b9a626c6d65 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 11 Mar 2020 20:45:54 -0300 Subject: [PATCH 254/313] [doc] Feed exports: full local path as example --- docs/topics/feed-exports.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index 6d6ba33c9..9e5968a29 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -249,7 +249,7 @@ For instance:: 'fields': None, 'indent': 4, }, - 'items.xml': { + '/home/user/documents/items.xml': { 'format': 'xml', 'fields': ['name', 'price'], 'encoding': 'latin1', From f3bab819ab92cc0750c9b73141abdcbc8da7c4ac Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 11 Mar 2020 20:56:25 -0300 Subject: [PATCH 255/313] Add tests for scrapy.utils.conf.feed_complete_default_values_from_settings --- tests/test_utils_conf.py | 45 +++++++++++++++++++++++++++++++++++++++- 1 file changed, 44 insertions(+), 1 deletion(-) diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index f064a646c..332120021 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -3,7 +3,12 @@ import warnings from scrapy.exceptions import UsageError, ScrapyDeprecationWarning from scrapy.settings import BaseSettings, Settings -from scrapy.utils.conf import build_component_list, arglist_to_dict, feed_process_params_from_cli +from scrapy.utils.conf import ( + arglist_to_dict, + build_component_list, + feed_complete_default_values_from_settings, + feed_process_params_from_cli +) class BuildComponentListTest(unittest.TestCase): @@ -135,6 +140,44 @@ class FeedExportConfigTestCase(unittest.TestCase): feed_process_params_from_cli(settings, ['-:pickle']) ) + def test_feed_complete_default_values_from_settings_empty(self): + feed = {} + settings = Settings({ + "FEED_EXPORT_ENCODING": "custom encoding", + "FEED_EXPORT_FIELDS": ["f1", "f2", "f3"], + "FEED_EXPORT_INDENT": 42, + "FEED_STORE_EMPTY": True, + "FEED_URI_PARAMS": (1, 2, 3, 4), + }) + new_feed = feed_complete_default_values_from_settings(feed, settings) + self.assertEqual(new_feed, { + "encoding": "custom encoding", + "fields": ["f1", "f2", "f3"], + "indent": 42, + "store_empty": True, + "uri_params": (1, 2, 3, 4), + }) + + def test_feed_complete_default_values_from_settings_non_empty(self): + feed = { + "encoding": "other encoding", + "fields": None, + } + settings = Settings({ + "FEED_EXPORT_ENCODING": "custom encoding", + "FEED_EXPORT_FIELDS": ["f1", "f2", "f3"], + "FEED_EXPORT_INDENT": 42, + "FEED_STORE_EMPTY": True, + }) + new_feed = feed_complete_default_values_from_settings(feed, settings) + self.assertEqual(new_feed, { + "encoding": "other encoding", + "fields": None, + "indent": 42, + "store_empty": True, + "uri_params": None, + }) + if __name__ == "__main__": unittest.main() From c886a70eae35e628040da0504af3fd9aaa6aea75 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 11 Mar 2020 21:06:51 -0300 Subject: [PATCH 256/313] Use dict.setdefault in scrapy.utils.conf.feed_complete_default_values_from_settings --- scrapy/utils/conf.py | 18 ++++++++---------- 1 file changed, 8 insertions(+), 10 deletions(-) diff --git a/scrapy/utils/conf.py b/scrapy/utils/conf.py index e01027491..5921f82bf 100644 --- a/scrapy/utils/conf.py +++ b/scrapy/utils/conf.py @@ -113,16 +113,14 @@ def get_sources(use_closest=True): def feed_complete_default_values_from_settings(feed, settings): out = feed.copy() - if 'encoding' not in out: - out['encoding'] = settings['FEED_EXPORT_ENCODING'] - if 'fields' not in out: - out['fields'] = settings.getlist('FEED_EXPORT_FIELDS') or None - if 'indent' not in out: - out['indent'] = None if settings['FEED_EXPORT_INDENT'] is None else settings.getint('FEED_EXPORT_INDENT') - if 'store_empty' not in out: - out['store_empty'] = settings.getbool('FEED_STORE_EMPTY') - if 'uri_params' not in out: - out['uri_params'] = settings['FEED_URI_PARAMS'] + out.setdefault("encoding", settings["FEED_EXPORT_ENCODING"]) + out.setdefault("fields", settings.getlist("FEED_EXPORT_FIELDS") or None) + out.setdefault("store_empty", settings.getbool("FEED_STORE_EMPTY")) + out.setdefault("uri_params", settings["FEED_URI_PARAMS"]) + if settings["FEED_EXPORT_INDENT"] is None: + out.setdefault("indent", None) + else: + out.setdefault("indent", settings.getint("FEED_EXPORT_INDENT")) return out From 8d30dc08882e3b97dbaa17b3254de708c358d0d2 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Thu, 12 Mar 2020 09:36:15 -0300 Subject: [PATCH 257/313] Response.follow_all: return empty generators for empty sequences --- scrapy/http/response/text.py | 8 +++++--- tests/test_http_response.py | 4 ++++ 2 files changed, 9 insertions(+), 3 deletions(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 33a485328..2f0f3820c 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -188,9 +188,11 @@ class TextResponse(Response): selectors from which links cannot be obtained (for instance, anchor tags without an ``href`` attribute) """ - arg_count = len(list(filter(None, (urls, css, xpath)))) - if arg_count != 1: - raise ValueError('Please supply exactly one of the following arguments: urls, css, xpath') + arguments = [x for x in (urls, css, xpath) if x is not None] + if len(arguments) != 1: + raise ValueError( + "Please supply exactly one of the following arguments: urls, css, xpath" + ) if not urls: if css: urls = self.css(css) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index be17dfd6b..eafc3560e 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -215,6 +215,10 @@ class BaseResponseTest(unittest.TestCase): links = map(Link, absolute) self._assert_followed_all_urls(links, absolute) + def test_follow_all_empty(self): + r = self.response_class("http://example.com") + self.assertEqual([], list(r.follow_all([]))) + def test_follow_all_invalid(self): r = self.response_class("http://example.com") if self.response_class == Response: From 3b0820d747e11a1a7722f0777baf923607fc0485 Mon Sep 17 00:00:00 2001 From: nyov <nyov@users.noreply.github.com> Date: Thu, 12 Mar 2020 19:15:49 +0000 Subject: [PATCH 258/313] Deprecate Spider.make_requests_from_url, part 2 (#4412) --- scrapy/spiders/__init__.py | 6 ++++++ tests/test_spider.py | 8 +++++++- 2 files changed, 13 insertions(+), 1 deletion(-) diff --git a/scrapy/spiders/__init__.py b/scrapy/spiders/__init__.py index 9429f6cb2..ba1c866f8 100644 --- a/scrapy/spiders/__init__.py +++ b/scrapy/spiders/__init__.py @@ -78,6 +78,12 @@ class Spider(object_ref): def make_requests_from_url(self, url): """ This method is deprecated. """ + warnings.warn( + "Spider.make_requests_from_url method is deprecated: " + "it will be removed and not be called by the default " + "Spider.start_requests method in future Scrapy releases. " + "Please override Spider.start_requests method instead." + ) return Request(url, dont_filter=True) def parse(self, response): diff --git a/tests/test_spider.py b/tests/test_spider.py index 317a27076..bb00c8f42 100644 --- a/tests/test_spider.py +++ b/tests/test_spider.py @@ -602,13 +602,19 @@ class DeprecationTest(unittest.TestCase): self.assertEqual(len(list(spider1.start_requests())), 1) self.assertEqual(len(w), 0) + # spider without overridden make_requests_from_url method + # should issue a warning when called directly + request = spider1.make_requests_from_url("http://www.example.com") + self.assertTrue(isinstance(request, Request)) + self.assertEqual(len(w), 1) + # spider with overridden make_requests_from_url issues a warning, # but the method still works spider2 = MySpider5() requests = list(spider2.start_requests()) self.assertEqual(len(requests), 1) self.assertEqual(requests[0].url, 'http://example.com/foo') - self.assertEqual(len(w), 1) + self.assertEqual(len(w), 2) class NoParseMethodSpiderTest(unittest.TestCase): From ccc4d88779cf2827431ef9e73f976f755c04fe0e Mon Sep 17 00:00:00 2001 From: Lukas Anzinger <lukas@lukasanzinger.at> Date: Thu, 12 Mar 2020 20:42:14 +0100 Subject: [PATCH 259/313] Ignore a domain in allowed_domains with port and issue a warning (#4413) --- scrapy/spidermiddlewares/offsite.py | 11 ++++++++++- tests/test_spidermiddleware_offsite.py | 15 +++++++++++++-- 2 files changed, 23 insertions(+), 3 deletions(-) diff --git a/scrapy/spidermiddlewares/offsite.py b/scrapy/spidermiddlewares/offsite.py index 36f809699..2fab572e6 100644 --- a/scrapy/spidermiddlewares/offsite.py +++ b/scrapy/spidermiddlewares/offsite.py @@ -53,7 +53,8 @@ class OffsiteMiddleware(object): allowed_domains = getattr(spider, 'allowed_domains', None) if not allowed_domains: return re.compile('') # allow all by default - url_pattern = re.compile("^https?://.*$") + url_pattern = re.compile(r"^https?://.*$") + port_pattern = re.compile(r":\d+$") domains = [] for domain in allowed_domains: if domain is None: @@ -62,6 +63,10 @@ class OffsiteMiddleware(object): message = ("allowed_domains accepts only domains, not URLs. " "Ignoring URL entry %s in allowed_domains." % domain) warnings.warn(message, URLWarning) + elif port_pattern.search(domain): + message = ("allowed_domains accepts only domains without ports. " + "Ignoring entry %s in allowed_domains." % domain) + warnings.warn(message, PortWarning) else: domains.append(re.escape(domain)) regex = r'^(.*\.)?(%s)$' % '|'.join(domains) @@ -74,3 +79,7 @@ class OffsiteMiddleware(object): class URLWarning(Warning): pass + + +class PortWarning(Warning): + pass diff --git a/tests/test_spidermiddleware_offsite.py b/tests/test_spidermiddleware_offsite.py index 51c328943..b96807bc2 100644 --- a/tests/test_spidermiddleware_offsite.py +++ b/tests/test_spidermiddleware_offsite.py @@ -4,7 +4,7 @@ import warnings from scrapy.http import Response, Request from scrapy.spiders import Spider -from scrapy.spidermiddlewares.offsite import OffsiteMiddleware, URLWarning +from scrapy.spidermiddlewares.offsite import OffsiteMiddleware, URLWarning, PortWarning from scrapy.utils.test import get_crawler @@ -26,7 +26,8 @@ class TestOffsiteMiddleware(TestCase): Request('http://scrapy.org/1'), Request('http://sub.scrapy.org/1'), Request('http://offsite.tld/letmepass', dont_filter=True), - Request('http://scrapy.test.org/')] + Request('http://scrapy.test.org/'), + Request('http://scrapy.test.org:8000/')] offsite_reqs = [Request('http://scrapy2.org'), Request('http://offsite.tld/'), Request('http://offsite.tld/scrapytest.org'), @@ -80,3 +81,13 @@ class TestOffsiteMiddleware5(TestOffsiteMiddleware4): warnings.simplefilter("always") self.mw.get_host_regex(self.spider) assert issubclass(w[-1].category, URLWarning) + + +class TestOffsiteMiddleware6(TestOffsiteMiddleware4): + + def test_get_host_regex(self): + self.spider.allowed_domains = ['scrapytest.org:8000', 'scrapy.org', 'scrapy.test.org'] + with warnings.catch_warnings(record=True) as w: + warnings.simplefilter("always") + self.mw.get_host_regex(self.spider) + assert issubclass(w[-1].category, PortWarning) From 3f6cdcabceff5c5b2ac2935f8073b6efe817645b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Fri, 13 Mar 2020 13:25:53 +0100 Subject: [PATCH 260/313] Restrict pytest to versions prior to 5.4 --- tests/requirements-py3.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index d97c4b8ee..d207c5fb0 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -2,7 +2,7 @@ jmespath mitmproxy; python_version >= '3.6' mitmproxy<4.0.0; python_version < '3.6' -pytest +pytest < 5.4 pytest-cov pytest-twisted >= 1.11 pytest-xdist From f9bf4b8d4dd64a1d65e949927b8ea7ad34e756d3 Mon Sep 17 00:00:00 2001 From: Aditya Kumar <k.aditya00@gmail.com> Date: Sat, 14 Mar 2020 15:09:00 +0530 Subject: [PATCH 261/313] Remove all top-level imports for twisted.internet.reactor (#4406) --- docs/topics/settings.rst | 61 ++++++++++++++++++++++- scrapy/core/downloader/__init__.py | 3 +- scrapy/core/downloader/handlers/ftp.py | 2 +- scrapy/core/downloader/handlers/http10.py | 3 +- scrapy/core/downloader/handlers/http11.py | 6 ++- scrapy/extensions/closespider.py | 3 +- scrapy/mail.py | 5 +- scrapy/utils/benchserver.py | 2 +- scrapy/utils/testproc.py | 3 +- scrapy/utils/testsite.py | 3 +- 10 files changed, 78 insertions(+), 13 deletions(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index a70023efa..c01202a10 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1473,7 +1473,66 @@ If a reactor is already installed, :meth:`CrawlerRunner.__init__ <scrapy.crawler.CrawlerRunner.__init__>` raises :exc:`Exception` if the installed reactor does not match the -:setting:`TWISTED_REACTOR` setting. +:setting:`TWISTED_REACTOR` setting; therfore, having top-level +:mod:`~twisted.internet.reactor` imports in project files and imported +third-party libraries will make Scrapy raise :exc:`Exception` when +it checks which reactor is installed. + +In order to use the reactor installed by Scrapy:: + + import scrapy + from twisted.internet import reactor + + + class QuotesSpider(scrapy.Spider): + name = 'quotes' + + def __init__(self, *args, **kwargs): + self.timeout = int(kwargs.pop('timeout', '60')) + super(QuotesSpider, self).__init__(*args, **kwargs) + + def start_requests(self): + reactor.callLater(self.timeout, self.stop) + + urls = ['http://quotes.toscrape.com/page/1'] + for url in urls: + yield scrapy.Request(url=url, callback=self.parse) + + def parse(self, response): + for quote in response.css('div.quote'): + yield {'text': quote.css('span.text::text').get()} + + def stop(self): + self.crawler.engine.close_spider(self, 'timeout') + + +which raises :exc:`Exception`, becomes:: + + import scrapy + + + class QuotesSpider(scrapy.Spider): + name = 'quotes' + + def __init__(self, *args, **kwargs): + self.timeout = int(kwargs.pop('timeout', '60')) + super(QuotesSpider, self).__init__(*args, **kwargs) + + def start_requests(self): + from twisted.internet import reactor + reactor.callLater(self.timeout, self.stop) + + urls = ['http://quotes.toscrape.com/page/1'] + for url in urls: + yield scrapy.Request(url=url, callback=self.parse) + + def parse(self, response): + for quote in response.css('div.quote'): + yield {'text': quote.css('span.text::text').get()} + + def stop(self): + self.crawler.engine.close_spider(self, 'timeout') + The default value of the :setting:`TWISTED_REACTOR` setting is ``None``, which means that Scrapy will not attempt to install any specific reactor, and the diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index 5a2fdadf5..644be121f 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -3,7 +3,7 @@ from time import time from datetime import datetime from collections import deque -from twisted.internet import reactor, defer, task +from twisted.internet import defer, task from scrapy.utils.defer import mustbe_deferred from scrapy.utils.httpobj import urlparse_cached @@ -133,6 +133,7 @@ class Downloader(object): return deferred def _process_queue(self, spider, slot): + from twisted.internet import reactor if slot.latercall and slot.latercall.active(): return diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 1681c6df8..432cb1831 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -32,7 +32,6 @@ import re from io import BytesIO from urllib.parse import unquote -from twisted.internet import reactor from twisted.internet.protocol import ClientCreator, Protocol from twisted.protocols.ftp import CommandFailed, FTPClient @@ -81,6 +80,7 @@ class FTPDownloadHandler: return cls(crawler.settings) def download_request(self, request, spider): + from twisted.internet import reactor parsed_url = urlparse_cached(request) user = request.meta.get("ftp_user", self.default_user) password = request.meta.get("ftp_password", self.default_password) diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index d4aa51bd1..c0146a0a6 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -1,7 +1,5 @@ """Download handlers for http and https schemes """ -from twisted.internet import reactor - from scrapy.utils.misc import create_instance, load_object from scrapy.utils.python import to_unicode @@ -26,6 +24,7 @@ class HTTP10DownloadHandler: return factory.deferred def _connect(self, factory): + from twisted.internet import reactor host, port = to_unicode(factory.host), factory.port if factory.scheme == b'https': client_context_factory = create_instance( diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 93951d3b5..04a8d617a 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -8,7 +8,7 @@ from io import BytesIO from time import time from urllib.parse import urldefrag -from twisted.internet import defer, protocol, reactor, ssl +from twisted.internet import defer, protocol, ssl from twisted.internet.endpoints import TCP4ClientEndpoint from twisted.internet.error import TimeoutError from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI @@ -33,6 +33,7 @@ class HTTP11DownloadHandler: lazy = False def __init__(self, settings, crawler=None): + from twisted.internet import reactor self._pool = HTTPConnectionPool(reactor, persistent=True) self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN') self._pool._factory.noisy = False @@ -81,6 +82,7 @@ class HTTP11DownloadHandler: return agent.download_request(request) def close(self): + from twisted.internet import reactor d = self._pool.closeCachedConnections() # closeCachedConnections will hang on network or server issues, so # we'll manually timeout the deferred. @@ -284,6 +286,7 @@ class ScrapyAgent(object): self._txresponse = None def _get_agent(self, request, timeout): + from twisted.internet import reactor bindaddress = request.meta.get('bindaddress') or self._bindAddress proxy = request.meta.get('proxy') if proxy: @@ -326,6 +329,7 @@ class ScrapyAgent(object): ) def download_request(self, request): + from twisted.internet import reactor timeout = request.meta.get('download_timeout') or self._connectTimeout agent = self._get_agent(request, timeout) diff --git a/scrapy/extensions/closespider.py b/scrapy/extensions/closespider.py index afb2ed049..260b2e86e 100644 --- a/scrapy/extensions/closespider.py +++ b/scrapy/extensions/closespider.py @@ -6,8 +6,6 @@ See documentation in docs/topics/extensions.rst from collections import defaultdict -from twisted.internet import reactor - from scrapy import signals from scrapy.exceptions import NotConfigured @@ -54,6 +52,7 @@ class CloseSpider(object): self.crawler.engine.close_spider(spider, 'closespider_pagecount') def spider_opened(self, spider): + from twisted.internet import reactor self.task = reactor.callLater(self.close_on['timeout'], self.crawler.engine.close_spider, spider, reason='closespider_timeout') diff --git a/scrapy/mail.py b/scrapy/mail.py index 9655b8114..b2a24a3db 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -12,7 +12,7 @@ from email.mime.text import MIMEText from email.utils import COMMASPACE, formatdate from io import BytesIO -from twisted.internet import defer, reactor, ssl +from twisted.internet import defer, ssl from scrapy.utils.misc import arg_to_iter from scrapy.utils.python import to_bytes @@ -28,7 +28,6 @@ def _to_bytes_or_none(text): class MailSender(object): - def __init__(self, smtphost='localhost', mailfrom='scrapy@localhost', smtpuser=None, smtppass=None, smtpport=25, smtptls=False, smtpssl=False, debug=False): self.smtphost = smtphost @@ -47,6 +46,7 @@ class MailSender(object): settings.getbool('MAIL_TLS'), settings.getbool('MAIL_SSL')) def send(self, to, subject, body, cc=None, attachs=(), mimetype='text/plain', charset=None, _callback=None): + from twisted.internet import reactor if attachs: msg = MIMEMultipart() else: @@ -111,6 +111,7 @@ class MailSender(object): def _sendmail(self, to_addrs, msg): # Import twisted.mail here because it is not available in python3 + from twisted.internet import reactor from twisted.mail.smtp import ESMTPSenderFactory msg = BytesIO(msg) d = defer.Deferred() diff --git a/scrapy/utils/benchserver.py b/scrapy/utils/benchserver.py index cdbe21942..9d8d64612 100644 --- a/scrapy/utils/benchserver.py +++ b/scrapy/utils/benchserver.py @@ -3,7 +3,6 @@ from urllib.parse import urlencode from twisted.web.server import Site from twisted.web.resource import Resource -from twisted.internet import reactor class Root(Resource): @@ -34,6 +33,7 @@ def _getarg(request, name, default=None, type=str): if __name__ == '__main__': + from twisted.internet import reactor root = Root() factory = Site(root) httpPort = reactor.listenTCP(8998, Site(root)) diff --git a/scrapy/utils/testproc.py b/scrapy/utils/testproc.py index 0f15cf60a..37803b287 100644 --- a/scrapy/utils/testproc.py +++ b/scrapy/utils/testproc.py @@ -1,7 +1,7 @@ import sys import os -from twisted.internet import reactor, defer, protocol +from twisted.internet import defer, protocol class ProcessTest(object): @@ -11,6 +11,7 @@ class ProcessTest(object): cwd = os.getcwd() # trial chdirs to temp dir def execute(self, args, check_code=True, settings=None): + from twisted.internet import reactor env = os.environ.copy() if settings is not None: env['SCRAPY_SETTINGS_MODULE'] = settings diff --git a/scrapy/utils/testsite.py b/scrapy/utils/testsite.py index 6f5c21624..9e1598805 100644 --- a/scrapy/utils/testsite.py +++ b/scrapy/utils/testsite.py @@ -1,12 +1,12 @@ from urllib.parse import urljoin -from twisted.internet import reactor from twisted.web import server, resource, static, util class SiteTest(object): def setUp(self): + from twisted.internet import reactor super(SiteTest, self).setUp() self.site = reactor.listenTCP(0, test_site(), interface="127.0.0.1") self.baseurl = "http://localhost:%d/" % self.site.getHost().port @@ -38,6 +38,7 @@ def test_site(): if __name__ == '__main__': + from twisted.internet import reactor port = reactor.listenTCP(0, test_site(), interface="127.0.0.1") print("http://localhost:%d/" % port.getHost().port) reactor.run() From e5711127b162ce78e6c1b60ef209792cd6eef4e6 Mon Sep 17 00:00:00 2001 From: elacuesta <elacuesta@users.noreply.github.com> Date: Mon, 16 Mar 2020 15:43:02 -0300 Subject: [PATCH 262/313] Remove deprecated ChunkedTransferMiddleware (#4431) --- scrapy/downloadermiddlewares/chunked.py | 21 --------------------- 1 file changed, 21 deletions(-) delete mode 100644 scrapy/downloadermiddlewares/chunked.py diff --git a/scrapy/downloadermiddlewares/chunked.py b/scrapy/downloadermiddlewares/chunked.py deleted file mode 100644 index 6748d0265..000000000 --- a/scrapy/downloadermiddlewares/chunked.py +++ /dev/null @@ -1,21 +0,0 @@ -import warnings - -from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.http import decode_chunked_transfer - - -warnings.warn("Module `scrapy.downloadermiddlewares.chunked` is deprecated, " - "chunked transfers are supported by default.", - ScrapyDeprecationWarning, stacklevel=2) - - -class ChunkedTransferMiddleware(object): - """This middleware adds support for chunked transfer encoding, as - documented in: https://en.wikipedia.org/wiki/Chunked_transfer_encoding - """ - - def process_response(self, request, response, spider): - if response.headers.get('Transfer-Encoding') == 'chunked': - body = decode_chunked_transfer(response.body) - return response.replace(body=body) - return response From dfbe1d95071acfbba159cea051530749b1684460 Mon Sep 17 00:00:00 2001 From: elacuesta <elacuesta@users.noreply.github.com> Date: Mon, 16 Mar 2020 16:12:46 -0300 Subject: [PATCH 263/313] Remove object base class (#4430) --- docs/topics/exporters.rst | 2 +- docs/topics/extensions.rst | 2 +- docs/topics/item-pipeline.rst | 10 +++++----- docs/topics/leaks.rst | 2 +- docs/topics/settings.rst | 2 +- docs/topics/stats.rst | 2 +- scrapy/commands/__init__.py | 2 +- scrapy/commands/bench.py | 2 +- scrapy/contracts/__init__.py | 4 ++-- scrapy/core/downloader/__init__.py | 4 ++-- scrapy/core/downloader/handlers/http11.py | 4 ++-- scrapy/core/engine.py | 4 ++-- scrapy/core/scheduler.py | 2 +- scrapy/core/scraper.py | 4 ++-- scrapy/downloadermiddlewares/ajaxcrawl.py | 2 +- scrapy/downloadermiddlewares/cookies.py | 2 +- scrapy/downloadermiddlewares/decompression.py | 2 +- scrapy/downloadermiddlewares/defaultheaders.py | 2 +- .../downloadermiddlewares/downloadtimeout.py | 2 +- scrapy/downloadermiddlewares/httpauth.py | 2 +- scrapy/downloadermiddlewares/httpcache.py | 2 +- .../downloadermiddlewares/httpcompression.py | 2 +- scrapy/downloadermiddlewares/httpproxy.py | 2 +- scrapy/downloadermiddlewares/redirect.py | 2 +- scrapy/downloadermiddlewares/retry.py | 2 +- scrapy/downloadermiddlewares/robotstxt.py | 2 +- scrapy/downloadermiddlewares/stats.py | 2 +- scrapy/downloadermiddlewares/useragent.py | 2 +- scrapy/dupefilters.py | 2 +- scrapy/exporters.py | 2 +- scrapy/extensions/closespider.py | 2 +- scrapy/extensions/corestats.py | 2 +- scrapy/extensions/debug.py | 4 ++-- scrapy/extensions/feedexport.py | 10 +++++----- scrapy/extensions/httpcache.py | 8 ++++---- scrapy/extensions/logstats.py | 2 +- scrapy/extensions/memdebug.py | 2 +- scrapy/extensions/memusage.py | 2 +- scrapy/extensions/spiderstate.py | 2 +- scrapy/extensions/statsmailer.py | 2 +- scrapy/extensions/throttle.py | 2 +- scrapy/http/cookies.py | 8 ++++---- scrapy/link.py | 2 +- scrapy/linkextractors/__init__.py | 2 +- scrapy/linkextractors/lxmlhtml.py | 2 +- scrapy/loader/__init__.py | 2 +- scrapy/loader/processors.py | 12 ++++++------ scrapy/logformatter.py | 2 +- scrapy/mail.py | 2 +- scrapy/middleware.py | 2 +- scrapy/pipelines/files.py | 8 ++++---- scrapy/pipelines/media.py | 4 ++-- scrapy/pqueues.py | 6 +++--- scrapy/responsetypes.py | 2 +- scrapy/settings/__init__.py | 2 +- scrapy/shell.py | 2 +- scrapy/signalmanager.py | 2 +- scrapy/spiderloader.py | 2 +- scrapy/spidermiddlewares/depth.py | 2 +- scrapy/spidermiddlewares/httperror.py | 2 +- scrapy/spidermiddlewares/offsite.py | 2 +- scrapy/spidermiddlewares/referer.py | 4 ++-- scrapy/spidermiddlewares/urllength.py | 2 +- scrapy/spiders/crawl.py | 2 +- scrapy/statscollectors.py | 2 +- .../project/module/middlewares.py.tmpl | 4 ++-- .../templates/project/module/pipelines.py.tmpl | 2 +- scrapy/utils/datatypes.py | 2 +- scrapy/utils/deprecate.py | 2 +- scrapy/utils/iterators.py | 2 +- scrapy/utils/log.py | 2 +- scrapy/utils/python.py | 4 ++-- scrapy/utils/reactor.py | 2 +- scrapy/utils/sitemap.py | 2 +- scrapy/utils/testproc.py | 2 +- scrapy/utils/testsite.py | 2 +- scrapy/utils/trackref.py | 5 ++--- tests/pipelines.py | 4 ++-- tests/test_cmdline/extensions.py | 4 ++-- .../test_spider/pipelines.py | 4 ++-- tests/test_command_parse.py | 2 +- tests/test_contracts.py | 2 +- tests/test_crawler.py | 2 +- tests/test_downloadermiddleware.py | 2 +- tests/test_dupefilters.py | 2 +- tests/test_engine.py | 2 +- tests/test_feedexport.py | 2 +- tests/test_item.py | 2 +- tests/test_loader.py | 2 +- tests/test_logformatter.py | 2 +- tests/test_middleware.py | 8 ++++---- tests/test_request_cb_kwargs.py | 4 ++-- tests/test_scheduler.py | 6 +++--- tests/test_spidermiddleware_referer.py | 18 +++++++++--------- tests/test_squeues.py | 2 +- tests/test_utils_deprecate.py | 6 +++--- tests/test_utils_python.py | 8 ++++---- tests/test_utils_reqser.py | 2 +- 98 files changed, 155 insertions(+), 156 deletions(-) diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index d411e2eed..e52682690 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -42,7 +42,7 @@ value of one of their fields:: from scrapy.exporters import XmlItemExporter - class PerYearXmlExportPipeline(object): + class PerYearXmlExportPipeline: """Distribute items across multiple XML files according to their 'year' field""" def open_spider(self, spider): diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index dc057f6b6..94fd2e36e 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -107,7 +107,7 @@ Here is the code of such extension:: logger = logging.getLogger(__name__) - class SpiderOpenCloseLogging(object): + class SpiderOpenCloseLogging: def __init__(self, item_count): self.item_count = item_count diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index 98e2506e5..533f84630 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -81,7 +81,7 @@ contain a price:: from scrapy.exceptions import DropItem - class PricePipeline(object): + class PricePipeline: vat_factor = 1.15 @@ -103,7 +103,7 @@ format:: import json - class JsonWriterPipeline(object): + class JsonWriterPipeline: def open_spider(self, spider): self.file = open('items.jl', 'w') @@ -132,7 +132,7 @@ method and how to clean up the resources properly.:: import pymongo - class MongoPipeline(object): + class MongoPipeline: collection_name = 'scrapy_items' @@ -180,7 +180,7 @@ it saves the screenshot to a file and adds filename to the item. from urllib.parse import quote - class ScreenshotPipeline(object): + class ScreenshotPipeline: """Pipeline that uses Splash to render screenshot of every Scrapy item.""" @@ -219,7 +219,7 @@ returns multiples items with the same id:: from scrapy.exceptions import DropItem - class DuplicatesPipeline(object): + class DuplicatesPipeline: def __init__(self): self.ids_seen = set() diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index c0c83fc84..4ee447065 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -170,7 +170,7 @@ Here are the functions available in the :mod:`~scrapy.utils.trackref` module. .. class:: object_ref - Inherit from this class (instead of object) if you want to track live + Inherit from this class if you want to track live instances with the ``trackref`` module. .. function:: print_live_refs(class_name, ignore=NoneType) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index c01202a10..dc6843d75 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -124,7 +124,7 @@ Settings can be accessed through the :attr:`scrapy.crawler.Crawler.settings` attribute of the Crawler that is passed to ``from_crawler`` method in extensions, middlewares and item pipelines:: - class MyExtension(object): + class MyExtension: def __init__(self, log_is_enabled=False): if log_is_enabled: print("log is enabled!") diff --git a/docs/topics/stats.rst b/docs/topics/stats.rst index 3dd829ebe..af848b402 100644 --- a/docs/topics/stats.rst +++ b/docs/topics/stats.rst @@ -32,7 +32,7 @@ Common Stats Collector uses Access the stats collector through the :attr:`~scrapy.crawler.Crawler.stats` attribute. Here is an example of an extension that access stats:: - class ExtensionThatAccessStats(object): + class ExtensionThatAccessStats: def __init__(self, stats): self.stats = stats diff --git a/scrapy/commands/__init__.py b/scrapy/commands/__init__.py index 0b24193c2..a573a03d9 100644 --- a/scrapy/commands/__init__.py +++ b/scrapy/commands/__init__.py @@ -9,7 +9,7 @@ from scrapy.utils.conf import arglist_to_dict from scrapy.exceptions import UsageError -class ScrapyCommand(object): +class ScrapyCommand: requires_project = False crawler_process = None diff --git a/scrapy/commands/bench.py b/scrapy/commands/bench.py index 7bbe362e7..c9f3b38e0 100644 --- a/scrapy/commands/bench.py +++ b/scrapy/commands/bench.py @@ -25,7 +25,7 @@ class Command(ScrapyCommand): self.crawler_process.start() -class _BenchServer(object): +class _BenchServer: def __enter__(self): from scrapy.utils.test import get_testenv diff --git a/scrapy/contracts/__init__.py b/scrapy/contracts/__init__.py index 7b6591d86..41d4f25b2 100644 --- a/scrapy/contracts/__init__.py +++ b/scrapy/contracts/__init__.py @@ -9,7 +9,7 @@ from scrapy.utils.spider import iterate_spider_output from scrapy.utils.python import get_spec -class ContractsManager(object): +class ContractsManager: contracts = {} def __init__(self, contracts): @@ -107,7 +107,7 @@ class ContractsManager(object): request.errback = eb_wrapper -class Contract(object): +class Contract: """ Abstract class for contracts """ request_cls = None diff --git a/scrapy/core/downloader/__init__.py b/scrapy/core/downloader/__init__.py index 644be121f..36aca4dae 100644 --- a/scrapy/core/downloader/__init__.py +++ b/scrapy/core/downloader/__init__.py @@ -13,7 +13,7 @@ from scrapy.core.downloader.middleware import DownloaderMiddlewareManager from scrapy.core.downloader.handlers import DownloadHandlers -class Slot(object): +class Slot: """Downloader slot""" def __init__(self, concurrency, delay, randomize_delay): @@ -66,7 +66,7 @@ def _get_concurrency_delay(concurrency, spider, settings): return concurrency, delay -class Downloader(object): +class Downloader: DOWNLOAD_SLOT = 'download_slot' diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 04a8d617a..c970909d7 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -268,7 +268,7 @@ class ScrapyProxyAgent(Agent): ) -class ScrapyAgent(object): +class ScrapyAgent: _Agent = Agent _ProxyAgent = ScrapyProxyAgent @@ -432,7 +432,7 @@ class ScrapyAgent(object): @implementer(IBodyProducer) -class _RequestBodyProducer(object): +class _RequestBodyProducer: def __init__(self, body): self.body = body diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 6ab8cde6b..74f03344e 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -21,7 +21,7 @@ from scrapy.utils.log import logformatter_adapter, failure_to_exc_info logger = logging.getLogger(__name__) -class Slot(object): +class Slot: def __init__(self, start_requests, close_if_idle, nextcall, scheduler): self.closing = False @@ -53,7 +53,7 @@ class Slot(object): self.closing.callback(None) -class ExecutionEngine(object): +class ExecutionEngine: def __init__(self, crawler, spider_closed_callback): self.crawler = crawler diff --git a/scrapy/core/scheduler.py b/scrapy/core/scheduler.py index c96b9b719..a18c26b17 100644 --- a/scrapy/core/scheduler.py +++ b/scrapy/core/scheduler.py @@ -14,7 +14,7 @@ from scrapy.utils.deprecate import ScrapyDeprecationWarning logger = logging.getLogger(__name__) -class Scheduler(object): +class Scheduler: """ Scrapy Scheduler. It allows to enqueue requests and then get a next request to download. Scheduler is also handling duplication diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 41f015017..3e4826216 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -21,7 +21,7 @@ from scrapy.core.spidermw import SpiderMiddlewareManager logger = logging.getLogger(__name__) -class Slot(object): +class Slot: """Scraper slot (one per running spider)""" MIN_RESPONSE_SIZE = 1024 @@ -62,7 +62,7 @@ class Slot(object): return self.active_size > self.max_active_size -class Scraper(object): +class Scraper: def __init__(self, crawler): self.slot = None diff --git a/scrapy/downloadermiddlewares/ajaxcrawl.py b/scrapy/downloadermiddlewares/ajaxcrawl.py index 16b046e99..ad7a81e6b 100644 --- a/scrapy/downloadermiddlewares/ajaxcrawl.py +++ b/scrapy/downloadermiddlewares/ajaxcrawl.py @@ -11,7 +11,7 @@ from scrapy.http import HtmlResponse logger = logging.getLogger(__name__) -class AjaxCrawlMiddleware(object): +class AjaxCrawlMiddleware: """ Handle 'AJAX crawlable' pages marked as crawlable via meta tag. For more info see https://developers.google.com/webmasters/ajax-crawling/docs/getting-started. diff --git a/scrapy/downloadermiddlewares/cookies.py b/scrapy/downloadermiddlewares/cookies.py index d8dabdf13..d57f04bc3 100644 --- a/scrapy/downloadermiddlewares/cookies.py +++ b/scrapy/downloadermiddlewares/cookies.py @@ -10,7 +10,7 @@ from scrapy.utils.python import to_unicode logger = logging.getLogger(__name__) -class CookiesMiddleware(object): +class CookiesMiddleware: """This middleware enables working with sites that need cookies""" def __init__(self, debug=False): diff --git a/scrapy/downloadermiddlewares/decompression.py b/scrapy/downloadermiddlewares/decompression.py index fcea38ef5..0fcf8fb8c 100644 --- a/scrapy/downloadermiddlewares/decompression.py +++ b/scrapy/downloadermiddlewares/decompression.py @@ -16,7 +16,7 @@ from scrapy.responsetypes import responsetypes logger = logging.getLogger(__name__) -class DecompressionMiddleware(object): +class DecompressionMiddleware: """ This middleware tries to recognise and extract the possibly compressed responses that may arrive. """ diff --git a/scrapy/downloadermiddlewares/defaultheaders.py b/scrapy/downloadermiddlewares/defaultheaders.py index 93fe97673..f67961881 100644 --- a/scrapy/downloadermiddlewares/defaultheaders.py +++ b/scrapy/downloadermiddlewares/defaultheaders.py @@ -7,7 +7,7 @@ See documentation in docs/topics/downloader-middleware.rst from scrapy.utils.python import without_none_values -class DefaultHeadersMiddleware(object): +class DefaultHeadersMiddleware: def __init__(self, headers): self._headers = headers diff --git a/scrapy/downloadermiddlewares/downloadtimeout.py b/scrapy/downloadermiddlewares/downloadtimeout.py index 18123cfce..d373a22df 100644 --- a/scrapy/downloadermiddlewares/downloadtimeout.py +++ b/scrapy/downloadermiddlewares/downloadtimeout.py @@ -7,7 +7,7 @@ See documentation in docs/topics/downloader-middleware.rst from scrapy import signals -class DownloadTimeoutMiddleware(object): +class DownloadTimeoutMiddleware: def __init__(self, timeout=180): self._timeout = timeout diff --git a/scrapy/downloadermiddlewares/httpauth.py b/scrapy/downloadermiddlewares/httpauth.py index 7aa7a62bc..089bf0d85 100644 --- a/scrapy/downloadermiddlewares/httpauth.py +++ b/scrapy/downloadermiddlewares/httpauth.py @@ -9,7 +9,7 @@ from w3lib.http import basic_auth_header from scrapy import signals -class HttpAuthMiddleware(object): +class HttpAuthMiddleware: """Set Basic HTTP Authorization header (http_user and http_pass spider class attributes)""" diff --git a/scrapy/downloadermiddlewares/httpcache.py b/scrapy/downloadermiddlewares/httpcache.py index 4e06f8236..6db57bd8b 100644 --- a/scrapy/downloadermiddlewares/httpcache.py +++ b/scrapy/downloadermiddlewares/httpcache.py @@ -17,7 +17,7 @@ from scrapy.exceptions import IgnoreRequest, NotConfigured from scrapy.utils.misc import load_object -class HttpCacheMiddleware(object): +class HttpCacheMiddleware: DOWNLOAD_EXCEPTIONS = (defer.TimeoutError, TimeoutError, DNSLookupError, ConnectionRefusedError, ConnectionDone, ConnectError, diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 0010b2a8f..727c41466 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -15,7 +15,7 @@ except ImportError: pass -class HttpCompressionMiddleware(object): +class HttpCompressionMiddleware: """This middleware allows compressed (gzip, deflate) traffic to be sent/received from web sites""" @classmethod diff --git a/scrapy/downloadermiddlewares/httpproxy.py b/scrapy/downloadermiddlewares/httpproxy.py index 814ce78fe..da89d3e9b 100644 --- a/scrapy/downloadermiddlewares/httpproxy.py +++ b/scrapy/downloadermiddlewares/httpproxy.py @@ -7,7 +7,7 @@ from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.python import to_bytes -class HttpProxyMiddleware(object): +class HttpProxyMiddleware: def __init__(self, auth_encoding='latin-1'): self.auth_encoding = auth_encoding diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 08cff8a55..09ee8377e 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -11,7 +11,7 @@ from scrapy.exceptions import IgnoreRequest, NotConfigured logger = logging.getLogger(__name__) -class BaseRedirectMiddleware(object): +class BaseRedirectMiddleware: enabled_setting = 'REDIRECT_ENABLED' diff --git a/scrapy/downloadermiddlewares/retry.py b/scrapy/downloadermiddlewares/retry.py index 7ab5b6e62..bbf5fca05 100644 --- a/scrapy/downloadermiddlewares/retry.py +++ b/scrapy/downloadermiddlewares/retry.py @@ -25,7 +25,7 @@ from scrapy.utils.python import global_object_name logger = logging.getLogger(__name__) -class RetryMiddleware(object): +class RetryMiddleware: # IOError is raised by the HttpCompression middleware when trying to # decompress an empty response diff --git a/scrapy/downloadermiddlewares/robotstxt.py b/scrapy/downloadermiddlewares/robotstxt.py index 251706c50..7f18b2bf2 100644 --- a/scrapy/downloadermiddlewares/robotstxt.py +++ b/scrapy/downloadermiddlewares/robotstxt.py @@ -16,7 +16,7 @@ from scrapy.utils.misc import load_object logger = logging.getLogger(__name__) -class RobotsTxtMiddleware(object): +class RobotsTxtMiddleware: DOWNLOAD_PRIORITY = 1000 def __init__(self, crawler): diff --git a/scrapy/downloadermiddlewares/stats.py b/scrapy/downloadermiddlewares/stats.py index ef0aafce0..46a2ad397 100644 --- a/scrapy/downloadermiddlewares/stats.py +++ b/scrapy/downloadermiddlewares/stats.py @@ -4,7 +4,7 @@ from scrapy.utils.response import response_httprepr from scrapy.utils.python import global_object_name -class DownloaderStats(object): +class DownloaderStats: def __init__(self, stats): self.stats = stats diff --git a/scrapy/downloadermiddlewares/useragent.py b/scrapy/downloadermiddlewares/useragent.py index d24750c69..3ee7bd129 100644 --- a/scrapy/downloadermiddlewares/useragent.py +++ b/scrapy/downloadermiddlewares/useragent.py @@ -3,7 +3,7 @@ from scrapy import signals -class UserAgentMiddleware(object): +class UserAgentMiddleware: """This middleware allows spiders to override the user_agent""" def __init__(self, user_agent='Scrapy'): diff --git a/scrapy/dupefilters.py b/scrapy/dupefilters.py index a36c8304f..d74c8ed36 100644 --- a/scrapy/dupefilters.py +++ b/scrapy/dupefilters.py @@ -5,7 +5,7 @@ from scrapy.utils.job import job_dir from scrapy.utils.request import referer_str, request_fingerprint -class BaseDupeFilter(object): +class BaseDupeFilter: @classmethod def from_settings(cls, settings): diff --git a/scrapy/exporters.py b/scrapy/exporters.py index 2e20a7180..0cb6cef98 100644 --- a/scrapy/exporters.py +++ b/scrapy/exporters.py @@ -21,7 +21,7 @@ __all__ = ['BaseItemExporter', 'PprintItemExporter', 'PickleItemExporter', 'JsonItemExporter', 'MarshalItemExporter'] -class BaseItemExporter(object): +class BaseItemExporter: def __init__(self, *, dont_fail=False, **kwargs): self._kwargs = kwargs diff --git a/scrapy/extensions/closespider.py b/scrapy/extensions/closespider.py index 260b2e86e..e3f212bef 100644 --- a/scrapy/extensions/closespider.py +++ b/scrapy/extensions/closespider.py @@ -10,7 +10,7 @@ from scrapy import signals from scrapy.exceptions import NotConfigured -class CloseSpider(object): +class CloseSpider: def __init__(self, crawler): self.crawler = crawler diff --git a/scrapy/extensions/corestats.py b/scrapy/extensions/corestats.py index 20adfbe4b..389cb65bc 100644 --- a/scrapy/extensions/corestats.py +++ b/scrapy/extensions/corestats.py @@ -6,7 +6,7 @@ from datetime import datetime from scrapy import signals -class CoreStats(object): +class CoreStats: def __init__(self, stats): self.stats = stats diff --git a/scrapy/extensions/debug.py b/scrapy/extensions/debug.py index 625e13249..586399784 100644 --- a/scrapy/extensions/debug.py +++ b/scrapy/extensions/debug.py @@ -17,7 +17,7 @@ from scrapy.utils.trackref import format_live_refs logger = logging.getLogger(__name__) -class StackTraceDump(object): +class StackTraceDump: def __init__(self, crawler=None): self.crawler = crawler @@ -52,7 +52,7 @@ class StackTraceDump(object): return dumps -class Debugger(object): +class Debugger: def __init__(self): try: signal.signal(signal.SIGUSR2, self._enter_debugger) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 108b6d35c..998d2a5d1 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -44,7 +44,7 @@ class IFeedStorage(Interface): @implementer(IFeedStorage) -class BlockingFeedStorage(object): +class BlockingFeedStorage: def open(self, spider): path = spider.crawler.settings['FEED_TEMPDIR'] @@ -61,7 +61,7 @@ class BlockingFeedStorage(object): @implementer(IFeedStorage) -class StdoutFeedStorage(object): +class StdoutFeedStorage: def __init__(self, uri, _stdout=None): if not _stdout: @@ -76,7 +76,7 @@ class StdoutFeedStorage(object): @implementer(IFeedStorage) -class FileFeedStorage(object): +class FileFeedStorage: def __init__(self, uri): self.path = file_uri_to_path(uri) @@ -179,7 +179,7 @@ class FTPFeedStorage(BlockingFeedStorage): ) -class _FeedSlot(object): +class _FeedSlot: def __init__(self, file, exporter, storage, uri, format, store_empty): self.file = file self.exporter = exporter @@ -203,7 +203,7 @@ class _FeedSlot(object): self._exporting = False -class FeedExporter(object): +class FeedExporter: @classmethod def from_crawler(cls, crawler): diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 91850683f..8546628a8 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -20,7 +20,7 @@ from scrapy.utils.request import request_fingerprint logger = logging.getLogger(__name__) -class DummyPolicy(object): +class DummyPolicy: def __init__(self, settings): self.ignore_schemes = settings.getlist('HTTPCACHE_IGNORE_SCHEMES') @@ -39,7 +39,7 @@ class DummyPolicy(object): return True -class RFC2616Policy(object): +class RFC2616Policy: MAXAGE = 3600 * 24 * 365 # one year @@ -213,7 +213,7 @@ class RFC2616Policy(object): return currentage -class DbmCacheStorage(object): +class DbmCacheStorage: def __init__(self, settings): self.cachedir = data_path(settings['HTTPCACHE_DIR'], createdir=True) @@ -270,7 +270,7 @@ class DbmCacheStorage(object): return request_fingerprint(request) -class FilesystemCacheStorage(object): +class FilesystemCacheStorage: def __init__(self, settings): self.cachedir = data_path(settings['HTTPCACHE_DIR']) diff --git a/scrapy/extensions/logstats.py b/scrapy/extensions/logstats.py index b685e7b19..0be2831a1 100644 --- a/scrapy/extensions/logstats.py +++ b/scrapy/extensions/logstats.py @@ -8,7 +8,7 @@ from scrapy import signals logger = logging.getLogger(__name__) -class LogStats(object): +class LogStats: """Log basic scraping stats periodically""" def __init__(self, stats, interval=60.0): diff --git a/scrapy/extensions/memdebug.py b/scrapy/extensions/memdebug.py index 892aa8a86..dc8cdbb1d 100644 --- a/scrapy/extensions/memdebug.py +++ b/scrapy/extensions/memdebug.py @@ -11,7 +11,7 @@ from scrapy.exceptions import NotConfigured from scrapy.utils.trackref import live_refs -class MemoryDebugger(object): +class MemoryDebugger: def __init__(self, stats): self.stats = stats diff --git a/scrapy/extensions/memusage.py b/scrapy/extensions/memusage.py index 14e0fb32d..a0540bf8f 100644 --- a/scrapy/extensions/memusage.py +++ b/scrapy/extensions/memusage.py @@ -19,7 +19,7 @@ from scrapy.utils.engine import get_engine_status logger = logging.getLogger(__name__) -class MemoryUsage(object): +class MemoryUsage: def __init__(self, crawler): if not crawler.settings.getbool('MEMUSAGE_ENABLED'): diff --git a/scrapy/extensions/spiderstate.py b/scrapy/extensions/spiderstate.py index 2c8e46914..2e5ff569f 100644 --- a/scrapy/extensions/spiderstate.py +++ b/scrapy/extensions/spiderstate.py @@ -6,7 +6,7 @@ from scrapy.exceptions import NotConfigured from scrapy.utils.job import job_dir -class SpiderState(object): +class SpiderState: """Store and load spider state during a scraping job""" def __init__(self, jobdir=None): diff --git a/scrapy/extensions/statsmailer.py b/scrapy/extensions/statsmailer.py index 6a982195d..320f13b29 100644 --- a/scrapy/extensions/statsmailer.py +++ b/scrapy/extensions/statsmailer.py @@ -8,7 +8,7 @@ from scrapy import signals from scrapy.mail import MailSender from scrapy.exceptions import NotConfigured -class StatsMailer(object): +class StatsMailer: def __init__(self, stats, recipients, mail): self.stats = stats diff --git a/scrapy/extensions/throttle.py b/scrapy/extensions/throttle.py index 198d4bbb0..56e5ad2d2 100644 --- a/scrapy/extensions/throttle.py +++ b/scrapy/extensions/throttle.py @@ -6,7 +6,7 @@ from scrapy import signals logger = logging.getLogger(__name__) -class AutoThrottle(object): +class AutoThrottle: def __init__(self, crawler): self.crawler = crawler diff --git a/scrapy/http/cookies.py b/scrapy/http/cookies.py index 0903fd4f8..3e810992c 100644 --- a/scrapy/http/cookies.py +++ b/scrapy/http/cookies.py @@ -5,7 +5,7 @@ from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.python import to_unicode -class CookieJar(object): +class CookieJar: def __init__(self, policy=None, check_expired_frequency=10000): self.policy = policy or DefaultCookiePolicy() self.jar = _CookieJar(self.policy) @@ -100,7 +100,7 @@ def potential_domain_matches(domain): return matches + ['.' + d for d in matches] -class _DummyLock(object): +class _DummyLock: def acquire(self): pass @@ -108,7 +108,7 @@ class _DummyLock(object): pass -class WrappedRequest(object): +class WrappedRequest: """Wraps a scrapy Request class with methods defined by urllib2.Request class to interact with CookieJar class see http://docs.python.org/library/urllib2.html#urllib2.Request @@ -178,7 +178,7 @@ class WrappedRequest(object): self.request.headers.appendlist(name, value) -class WrappedResponse(object): +class WrappedResponse: def __init__(self, response): self.response = response diff --git a/scrapy/link.py b/scrapy/link.py index a809c5ca4..7cb0765cc 100644 --- a/scrapy/link.py +++ b/scrapy/link.py @@ -6,7 +6,7 @@ its documentation in: docs/topics/link-extractors.rst """ -class Link(object): +class Link: """Link objects represent an extracted link by the LinkExtractor.""" __slots__ = ['url', 'text', 'fragment', 'nofollow'] diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index bdeab3a75..6afe867b5 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -49,7 +49,7 @@ _matches = lambda url, regexs: any(r.search(url) for r in regexs) _is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'} -class FilteringLinkExtractor(object): +class FilteringLinkExtractor: _csstranslator = HTMLTranslator() diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index ab82e1915..fbac1dc59 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -27,7 +27,7 @@ def _nons(tag): return tag -class LxmlParserLinkExtractor(object): +class LxmlParserLinkExtractor: def __init__(self, tag="a", attr="href", process=None, unique=False, strip=True, canonicalized=False): self.scan_tag = tag if callable(tag) else lambda t: t == tag diff --git a/scrapy/loader/__init__.py b/scrapy/loader/__init__.py index 7cf67e29e..21c4fb376 100644 --- a/scrapy/loader/__init__.py +++ b/scrapy/loader/__init__.py @@ -25,7 +25,7 @@ def unbound_method(method): return method -class ItemLoader(object): +class ItemLoader: default_item_class = Item default_input_processor = Identity() diff --git a/scrapy/loader/processors.py b/scrapy/loader/processors.py index 02c625acc..a7be65609 100644 --- a/scrapy/loader/processors.py +++ b/scrapy/loader/processors.py @@ -9,7 +9,7 @@ from scrapy.utils.misc import arg_to_iter from scrapy.loader.common import wrap_loader_context -class MapCompose(object): +class MapCompose: def __init__(self, *functions, **default_loader_context): self.functions = functions @@ -36,7 +36,7 @@ class MapCompose(object): return values -class Compose(object): +class Compose: def __init__(self, *functions, **default_loader_context): self.functions = functions @@ -61,7 +61,7 @@ class Compose(object): return value -class TakeFirst(object): +class TakeFirst: def __call__(self, values): for value in values: @@ -69,13 +69,13 @@ class TakeFirst(object): return value -class Identity(object): +class Identity: def __call__(self, values): return values -class SelectJmes(object): +class SelectJmes: """ Query the input string for the jmespath (given at instantiation), and return the answer @@ -95,7 +95,7 @@ class SelectJmes(object): return self.compiled_path.search(value) -class Join(object): +class Join: def __init__(self, separator=u' '): self.separator = separator diff --git a/scrapy/logformatter.py b/scrapy/logformatter.py index 14cec44a6..219145f13 100644 --- a/scrapy/logformatter.py +++ b/scrapy/logformatter.py @@ -14,7 +14,7 @@ DOWNLOADERRORMSG_SHORT = "Error downloading %(request)s" DOWNLOADERRORMSG_LONG = "Error downloading %(request)s: %(errmsg)s" -class LogFormatter(object): +class LogFormatter: """Class for generating log messages for different actions. All methods must return a dictionary listing the parameters ``level``, ``msg`` diff --git a/scrapy/mail.py b/scrapy/mail.py index b2a24a3db..9d186f4f3 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -27,7 +27,7 @@ def _to_bytes_or_none(text): return to_bytes(text) -class MailSender(object): +class MailSender: def __init__(self, smtphost='localhost', mailfrom='scrapy@localhost', smtpuser=None, smtppass=None, smtpport=25, smtptls=False, smtpssl=False, debug=False): self.smtphost = smtphost diff --git a/scrapy/middleware.py b/scrapy/middleware.py index 53fa435bb..5040378ea 100644 --- a/scrapy/middleware.py +++ b/scrapy/middleware.py @@ -9,7 +9,7 @@ from scrapy.utils.defer import process_parallel, process_chain, process_chain_bo logger = logging.getLogger(__name__) -class MiddlewareManager(object): +class MiddlewareManager: """Base class for implementing middleware managers""" component_name = 'foo middleware' diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 9b7445755..101bf5fbc 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -37,7 +37,7 @@ class FileException(Exception): """General media error exception""" -class FSFilesStore(object): +class FSFilesStore: def __init__(self, basedir): if '://' in basedir: basedir = basedir.split('://', 1)[1] @@ -75,7 +75,7 @@ class FSFilesStore(object): seen.add(dirname) -class S3FilesStore(object): +class S3FilesStore: AWS_ACCESS_KEY_ID = None AWS_SECRET_ACCESS_KEY = None AWS_ENDPOINT_URL = None @@ -213,7 +213,7 @@ class S3FilesStore(object): return extra -class GCSFilesStore(object): +class GCSFilesStore: GCS_PROJECT_ID = None @@ -259,7 +259,7 @@ class GCSFilesStore(object): ) -class FTPFilesStore(object): +class FTPFilesStore: FTP_USERNAME = None FTP_PASSWORD = None diff --git a/scrapy/pipelines/media.py b/scrapy/pipelines/media.py index c174addf9..562d9ee32 100644 --- a/scrapy/pipelines/media.py +++ b/scrapy/pipelines/media.py @@ -14,11 +14,11 @@ from scrapy.utils.log import failure_to_exc_info logger = logging.getLogger(__name__) -class MediaPipeline(object): +class MediaPipeline: LOG_FAILED_RESULTS = True - class SpiderInfo(object): + class SpiderInfo: def __init__(self, spider): self.spider = spider self.downloading = set() diff --git a/scrapy/pqueues.py b/scrapy/pqueues.py index 1afe58dab..e13d389ee 100644 --- a/scrapy/pqueues.py +++ b/scrapy/pqueues.py @@ -110,7 +110,7 @@ class ScrapyPriorityQueue: return sum(len(x) for x in self.queues.values()) if self.queues else 0 -class DownloaderInterface(object): +class DownloaderInterface: def __init__(self, crawler): self.downloader = crawler.engine.downloader @@ -129,8 +129,8 @@ class DownloaderInterface(object): return len(self.downloader.slots[slot].active) -class DownloaderAwarePriorityQueue(object): - """ PriorityQueue which takes Downloader activity in account: +class DownloaderAwarePriorityQueue: + """ PriorityQueue which takes Downloader activity into account: domains (slots) with the least amount of active downloads are dequeued first. """ diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 64bf93e86..ad89d9d22 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -11,7 +11,7 @@ from scrapy.utils.misc import load_object from scrapy.utils.python import binary_is_text, to_bytes, to_unicode -class ResponseTypes(object): +class ResponseTypes: CLASSES = { 'text/html': 'scrapy.http.HtmlResponse', diff --git a/scrapy/settings/__init__.py b/scrapy/settings/__init__.py index b6133619c..b9a13c018 100644 --- a/scrapy/settings/__init__.py +++ b/scrapy/settings/__init__.py @@ -28,7 +28,7 @@ def get_settings_priority(priority): return priority -class SettingsAttribute(object): +class SettingsAttribute: """Class for storing data related to settings attributes. diff --git a/scrapy/shell.py b/scrapy/shell.py index a5e140484..08ce89481 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -24,7 +24,7 @@ from scrapy.utils.conf import get_config from scrapy.utils.console import DEFAULT_PYTHON_SHELLS -class Shell(object): +class Shell: relevant_classes = (Crawler, Spider, Request, Response, BaseItem, Settings) diff --git a/scrapy/signalmanager.py b/scrapy/signalmanager.py index 481d97e9a..54eb7cfa3 100644 --- a/scrapy/signalmanager.py +++ b/scrapy/signalmanager.py @@ -2,7 +2,7 @@ from pydispatch import dispatcher from scrapy.utils import signal as _signal -class SignalManager(object): +class SignalManager: def __init__(self, sender=dispatcher.Anonymous): self.sender = sender diff --git a/scrapy/spiderloader.py b/scrapy/spiderloader.py index 048e84e4f..3be5aaec5 100644 --- a/scrapy/spiderloader.py +++ b/scrapy/spiderloader.py @@ -11,7 +11,7 @@ from scrapy.utils.spider import iter_spider_classes @implementer(ISpiderLoader) -class SpiderLoader(object): +class SpiderLoader: """ SpiderLoader is a class which locates and loads spiders in a Scrapy project. diff --git a/scrapy/spidermiddlewares/depth.py b/scrapy/spidermiddlewares/depth.py index 34a87f2df..fa7f5bef9 100644 --- a/scrapy/spidermiddlewares/depth.py +++ b/scrapy/spidermiddlewares/depth.py @@ -11,7 +11,7 @@ from scrapy.http import Request logger = logging.getLogger(__name__) -class DepthMiddleware(object): +class DepthMiddleware: def __init__(self, maxdepth, stats, verbose_stats=False, prio=1): self.maxdepth = maxdepth diff --git a/scrapy/spidermiddlewares/httperror.py b/scrapy/spidermiddlewares/httperror.py index def697c2b..375042340 100644 --- a/scrapy/spidermiddlewares/httperror.py +++ b/scrapy/spidermiddlewares/httperror.py @@ -18,7 +18,7 @@ class HttpError(IgnoreRequest): super(HttpError, self).__init__(*args, **kwargs) -class HttpErrorMiddleware(object): +class HttpErrorMiddleware: @classmethod def from_crawler(cls, crawler): diff --git a/scrapy/spidermiddlewares/offsite.py b/scrapy/spidermiddlewares/offsite.py index 2fab572e6..a006f3177 100644 --- a/scrapy/spidermiddlewares/offsite.py +++ b/scrapy/spidermiddlewares/offsite.py @@ -14,7 +14,7 @@ from scrapy.utils.httpobj import urlparse_cached logger = logging.getLogger(__name__) -class OffsiteMiddleware(object): +class OffsiteMiddleware: def __init__(self, stats): self.stats = stats diff --git a/scrapy/spidermiddlewares/referer.py b/scrapy/spidermiddlewares/referer.py index dce2b3598..3784de885 100644 --- a/scrapy/spidermiddlewares/referer.py +++ b/scrapy/spidermiddlewares/referer.py @@ -28,7 +28,7 @@ POLICY_UNSAFE_URL = "unsafe-url" POLICY_SCRAPY_DEFAULT = "scrapy-default" -class ReferrerPolicy(object): +class ReferrerPolicy: NOREFERRER_SCHEMES = LOCAL_SCHEMES @@ -284,7 +284,7 @@ def _load_policy_class(policy, warning_only=False): return None -class RefererMiddleware(object): +class RefererMiddleware: def __init__(self, settings=None): self.default_policy = DefaultReferrerPolicy diff --git a/scrapy/spidermiddlewares/urllength.py b/scrapy/spidermiddlewares/urllength.py index a904635d8..5be1f80cb 100644 --- a/scrapy/spidermiddlewares/urllength.py +++ b/scrapy/spidermiddlewares/urllength.py @@ -12,7 +12,7 @@ from scrapy.exceptions import NotConfigured logger = logging.getLogger(__name__) -class UrlLengthMiddleware(object): +class UrlLengthMiddleware: def __init__(self, maxlength): self.maxlength = maxlength diff --git a/scrapy/spiders/crawl.py b/scrapy/spiders/crawl.py index a2c364c0e..d76a96451 100644 --- a/scrapy/spiders/crawl.py +++ b/scrapy/spiders/crawl.py @@ -34,7 +34,7 @@ def _get_method(method, spider): _default_link_extractor = LinkExtractor() -class Rule(object): +class Rule: def __init__(self, link_extractor=None, callback=None, cb_kwargs=None, follow=None, process_links=None, process_request=None, errback=None): diff --git a/scrapy/statscollectors.py b/scrapy/statscollectors.py index f0bfaed34..579c60180 100644 --- a/scrapy/statscollectors.py +++ b/scrapy/statscollectors.py @@ -7,7 +7,7 @@ import logging logger = logging.getLogger(__name__) -class StatsCollector(object): +class StatsCollector: def __init__(self, crawler): self._dump = crawler.settings.getbool('STATS_DUMP') diff --git a/scrapy/templates/project/module/middlewares.py.tmpl b/scrapy/templates/project/module/middlewares.py.tmpl index 97b5db2e1..b3e58ff94 100644 --- a/scrapy/templates/project/module/middlewares.py.tmpl +++ b/scrapy/templates/project/module/middlewares.py.tmpl @@ -8,7 +8,7 @@ from scrapy import signals -class ${ProjectName}SpiderMiddleware(object): +class ${ProjectName}SpiderMiddleware: # Not all methods need to be defined. If a method is not defined, # scrapy acts as if the spider middleware does not modify the # passed objects. @@ -56,7 +56,7 @@ class ${ProjectName}SpiderMiddleware(object): spider.logger.info('Spider opened: %s' % spider.name) -class ${ProjectName}DownloaderMiddleware(object): +class ${ProjectName}DownloaderMiddleware: # Not all methods need to be defined. If a method is not defined, # scrapy acts as if the downloader middleware does not modify the # passed objects. diff --git a/scrapy/templates/project/module/pipelines.py.tmpl b/scrapy/templates/project/module/pipelines.py.tmpl index fb641d447..4876526a9 100644 --- a/scrapy/templates/project/module/pipelines.py.tmpl +++ b/scrapy/templates/project/module/pipelines.py.tmpl @@ -6,6 +6,6 @@ # See: https://docs.scrapy.org/en/latest/topics/item-pipeline.html -class ${ProjectName}Pipeline(object): +class ${ProjectName}Pipeline: def process_item(self, item, spider): return item diff --git a/scrapy/utils/datatypes.py b/scrapy/utils/datatypes.py index 175f92d77..f59f4cc55 100644 --- a/scrapy/utils/datatypes.py +++ b/scrapy/utils/datatypes.py @@ -109,7 +109,7 @@ class LocalWeakReferencedCache(weakref.WeakKeyDictionary): return None # key is not weak-referenceable, it's not cached -class SequenceExclude(object): +class SequenceExclude: """Object to test if an item is NOT within some sequence.""" def __init__(self, seq): diff --git a/scrapy/utils/deprecate.py b/scrapy/utils/deprecate.py index 2d3db431d..69334a918 100644 --- a/scrapy/utils/deprecate.py +++ b/scrapy/utils/deprecate.py @@ -144,7 +144,7 @@ def method_is_overridden(subclass, base_class, method_name): Return True if a method named ``method_name`` of a ``base_class`` is overridden in a ``subclass``. - >>> class Base(object): + >>> class Base: ... def foo(self): ... pass >>> class Sub1(Base): diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index 7849174fb..b71419111 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -52,7 +52,7 @@ def xmliter_lxml(obj, nodename, namespace=None, prefix='x'): yield xs.xpath(selxpath)[0] -class _StreamReader(object): +class _StreamReader: def __init__(self, obj): self._ptr = 0 diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index afef2c93f..5998dc33b 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -152,7 +152,7 @@ def log_scrapy_info(settings): logger.debug("Using reactor: %s.%s", reactor.__module__, reactor.__class__.__name__) -class StreamLogger(object): +class StreamLogger: """Fake file-like stream object that redirects writes to a logger instance Taken from: diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index e95a4648e..3d02d9478 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -223,7 +223,7 @@ def get_spec(func): >>> get_spec(re.match) (['pattern', 'string'], {'flags': 0}) - >>> class Test(object): + >>> class Test: ... def __call__(self, val): ... pass ... def method(self, val, flags=0): @@ -272,7 +272,7 @@ def equal_attributes(obj1, obj2, attributes): return True -class WeakKeyCache(object): +class WeakKeyCache: def __init__(self, default_factory): self.default_factory = default_factory diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 17d6b2857..5308812d6 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -24,7 +24,7 @@ def listen_tcp(portrange, host, factory): raise -class CallLaterOnce(object): +class CallLaterOnce: """Schedule a function to be called in the next reactor loop, but only if it hasn't been already scheduled since the last time it ran. """ diff --git a/scrapy/utils/sitemap.py b/scrapy/utils/sitemap.py index 2f10cf4de..a57a0c291 100644 --- a/scrapy/utils/sitemap.py +++ b/scrapy/utils/sitemap.py @@ -10,7 +10,7 @@ from urllib.parse import urljoin import lxml.etree -class Sitemap(object): +class Sitemap: """Class to parse Sitemap (type=urlset) and Sitemap Index (type=sitemapindex) files""" diff --git a/scrapy/utils/testproc.py b/scrapy/utils/testproc.py index 37803b287..a63c9a942 100644 --- a/scrapy/utils/testproc.py +++ b/scrapy/utils/testproc.py @@ -4,7 +4,7 @@ import os from twisted.internet import defer, protocol -class ProcessTest(object): +class ProcessTest: command = None prefix = [sys.executable, '-m', 'scrapy.cmdline'] diff --git a/scrapy/utils/testsite.py b/scrapy/utils/testsite.py index 9e1598805..66930ad2c 100644 --- a/scrapy/utils/testsite.py +++ b/scrapy/utils/testsite.py @@ -3,7 +3,7 @@ from urllib.parse import urljoin from twisted.web import server, resource, static, util -class SiteTest(object): +class SiteTest: def setUp(self): from twisted.internet import reactor diff --git a/scrapy/utils/trackref.py b/scrapy/utils/trackref.py index 4842b95df..baed5c536 100644 --- a/scrapy/utils/trackref.py +++ b/scrapy/utils/trackref.py @@ -19,9 +19,8 @@ NoneType = type(None) live_refs = defaultdict(weakref.WeakKeyDictionary) -class object_ref(object): - """Inherit from this class (instead of object) to a keep a record of live - instances""" +class object_ref: + """Inherit from this class to a keep a record of live instances""" __slots__ = () diff --git a/tests/pipelines.py b/tests/pipelines.py index de4894c32..cf677cc17 100644 --- a/tests/pipelines.py +++ b/tests/pipelines.py @@ -3,7 +3,7 @@ Some pipelines used for testing """ -class ZeroDivisionErrorPipeline(object): +class ZeroDivisionErrorPipeline: def open_spider(self, spider): a = 1 / 0 @@ -12,7 +12,7 @@ class ZeroDivisionErrorPipeline(object): return item -class ProcessWithZeroDivisionErrorPipiline(object): +class ProcessWithZeroDivisionErrorPipiline: def process_item(self, item, spider): 1 / 0 diff --git a/tests/test_cmdline/extensions.py b/tests/test_cmdline/extensions.py index c64e87d81..6504b4d2c 100644 --- a/tests/test_cmdline/extensions.py +++ b/tests/test_cmdline/extensions.py @@ -1,7 +1,7 @@ """A test extension used to check the settings loading order""" -class TestExtension(object): +class TestExtension: def __init__(self, settings): settings.set('TEST1', "%s + %s" % (settings['TEST1'], 'started')) @@ -11,5 +11,5 @@ class TestExtension(object): return cls(crawler.settings) -class DummyExtension(object): +class DummyExtension: pass diff --git a/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py b/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py index ce916f699..bd1f9cd8c 100644 --- a/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py +++ b/tests/test_cmdline_crawl_with_pipeline/test_spider/pipelines.py @@ -1,4 +1,4 @@ -class TestSpiderPipeline(object): +class TestSpiderPipeline: def open_spider(self, spider): pass @@ -7,7 +7,7 @@ class TestSpiderPipeline(object): return item -class TestSpiderExceptionPipeline(object): +class TestSpiderExceptionPipeline: def open_spider(self, spider): raise Exception('exception') diff --git a/tests/test_command_parse.py b/tests/test_command_parse.py index 5bf92b71a..85a24d0bc 100644 --- a/tests/test_command_parse.py +++ b/tests/test_command_parse.py @@ -89,7 +89,7 @@ class MyBadCrawlSpider(CrawlSpider): f.write(""" import logging -class MyPipeline(object): +class MyPipeline: component_name = 'my_pipeline' def process_item(self, item, spider): diff --git a/tests/test_contracts.py b/tests/test_contracts.py index 11d41c1fe..d1ce80f9d 100644 --- a/tests/test_contracts.py +++ b/tests/test_contracts.py @@ -25,7 +25,7 @@ class TestItem(Item): url = Field() -class ResponseMock(object): +class ResponseMock: url = 'http://scrapy.org' diff --git a/tests/test_crawler.py b/tests/test_crawler.py index 37a069611..169e763f0 100644 --- a/tests/test_crawler.py +++ b/tests/test_crawler.py @@ -126,7 +126,7 @@ class CrawlerLoggingTestCase(unittest.TestCase): self.assertEqual(crawler.stats.get_value('log_count/DEBUG', 0), 0) -class SpiderLoaderWithWrongInterface(object): +class SpiderLoaderWithWrongInterface: def unneeded_method(self): pass diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 3dd4f2351..a9190c62b 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -106,7 +106,7 @@ class ResponseFromProcessRequestTest(ManagerTestCase): def test_download_func_not_called(self): resp = Response('http://example.com/index.html') - class ResponseMiddleware(object): + class ResponseMiddleware: def process_request(self, request, spider): return resp diff --git a/tests/test_dupefilters.py b/tests/test_dupefilters.py index 9e24d86dd..ea0e664be 100644 --- a/tests/test_dupefilters.py +++ b/tests/test_dupefilters.py @@ -35,7 +35,7 @@ class FromSettingsRFPDupeFilter(RFPDupeFilter): return df -class DirectDupeFilter(object): +class DirectDupeFilter: method = 'n/a' diff --git a/tests/test_engine.py b/tests/test_engine.py index 25dee7c1f..5b7a4e676 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -96,7 +96,7 @@ def start_test_site(debug=False): return port -class CrawlerRun(object): +class CrawlerRun: """A class to run the crawler and keep track of events occurred""" def __init__(self, spider_class): diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index 08e8dfc41..c5589e52f 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -373,7 +373,7 @@ class StdoutFeedStorageTest(unittest.TestCase): self.assertEqual(out.getvalue(), b"content") -class FromCrawlerMixin(object): +class FromCrawlerMixin: init_with_crawler = False @classmethod diff --git a/tests/test_item.py b/tests/test_item.py index f70632d57..4017f6e84 100644 --- a/tests/test_item.py +++ b/tests/test_item.py @@ -213,7 +213,7 @@ class ItemTest(unittest.TestCase): class B(A): pass - class C(object): + class C: fields = {'load': Field(default='C')} not_allowed = Field(default='not_allowed') save = Field(default='C') diff --git a/tests/test_loader.py b/tests/test_loader.py index 579a85ff6..701d568dc 100644 --- a/tests/test_loader.py +++ b/tests/test_loader.py @@ -456,7 +456,7 @@ class BasicItemLoaderTest(unittest.TestCase): [u'marta', u'other'], Compose(float)) -class InitializationTestMixin(object): +class InitializationTestMixin: item_class = None diff --git a/tests/test_logformatter.py b/tests/test_logformatter.py index bf9fbe5e4..cd6cb8016 100644 --- a/tests/test_logformatter.py +++ b/tests/test_logformatter.py @@ -170,7 +170,7 @@ class SkipMessagesLogFormatter(LogFormatter): return None -class DropSomeItemsPipeline(object): +class DropSomeItemsPipeline: drop = True def process_item(self, item, spider): diff --git a/tests/test_middleware.py b/tests/test_middleware.py index ebf817c7e..3af514bb0 100644 --- a/tests/test_middleware.py +++ b/tests/test_middleware.py @@ -5,7 +5,7 @@ from scrapy.exceptions import NotConfigured from scrapy.middleware import MiddlewareManager -class M1(object): +class M1: def open_spider(self, spider): pass @@ -17,7 +17,7 @@ class M1(object): pass -class M2(object): +class M2: def open_spider(self, spider): pass @@ -28,13 +28,13 @@ class M2(object): pass -class M3(object): +class M3: def process(self, response, request, spider): pass -class MOff(object): +class MOff: def open_spider(self, spider): pass diff --git a/tests/test_request_cb_kwargs.py b/tests/test_request_cb_kwargs.py index a5cdc0de0..a3ddd50f4 100644 --- a/tests/test_request_cb_kwargs.py +++ b/tests/test_request_cb_kwargs.py @@ -8,7 +8,7 @@ from tests.spiders import MockServerSpider from tests.mockserver import MockServer -class InjectArgumentsDownloaderMiddleware(object): +class InjectArgumentsDownloaderMiddleware: """ Make sure downloader middlewares are able to update the keyword arguments """ @@ -23,7 +23,7 @@ class InjectArgumentsDownloaderMiddleware(object): return response -class InjectArgumentsSpiderMiddleware(object): +class InjectArgumentsSpiderMiddleware: """ Make sure spider middlewares are able to update the keyword arguments """ diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index 13c297084..00568aee9 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -20,7 +20,7 @@ MockEngine = collections.namedtuple('MockEngine', ['downloader']) MockSlot = collections.namedtuple('MockSlot', ['active']) -class MockDownloader(object): +class MockDownloader: def __init__(self): self.slots = dict() @@ -57,7 +57,7 @@ class MockCrawler(Crawler): self.engine = MockEngine(downloader=MockDownloader()) -class SchedulerHandler(object): +class SchedulerHandler: priority_queue_cls = None jobdir = None @@ -245,7 +245,7 @@ def _is_scheduling_fair(enqueued_slots, dequeued_slots): return True -class DownloaderAwareSchedulerTestMixin(object): +class DownloaderAwareSchedulerTestMixin: priority_queue_cls = 'scrapy.pqueues.DownloaderAwarePriorityQueue' reopen = False diff --git a/tests/test_spidermiddleware_referer.py b/tests/test_spidermiddleware_referer.py index 7cc17600c..4c6ede70b 100644 --- a/tests/test_spidermiddleware_referer.py +++ b/tests/test_spidermiddleware_referer.py @@ -47,7 +47,7 @@ class TestRefererMiddleware(TestCase): self.assertEqual(out[0].headers.get('Referer'), referrer) -class MixinDefault(object): +class MixinDefault: """ Based on https://www.w3.org/TR/referrer-policy/#referrer-policy-no-referrer-when-downgrade @@ -72,7 +72,7 @@ class MixinDefault(object): ] -class MixinNoReferrer(object): +class MixinNoReferrer: scenarii = [ ('https://example.com/page.html', 'https://example.com/', None), ('http://www.example.com/', 'https://scrapy.org/', None), @@ -82,7 +82,7 @@ class MixinNoReferrer(object): ] -class MixinNoReferrerWhenDowngrade(object): +class MixinNoReferrerWhenDowngrade: scenarii = [ # TLS to TLS: send non-empty referrer ('https://example.com/page.html', 'https://not.example.com/', b'https://example.com/page.html'), @@ -111,7 +111,7 @@ class MixinNoReferrerWhenDowngrade(object): ] -class MixinSameOrigin(object): +class MixinSameOrigin: scenarii = [ # Same origin (protocol, host, port): send referrer ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), @@ -144,7 +144,7 @@ class MixinSameOrigin(object): ] -class MixinOrigin(object): +class MixinOrigin: scenarii = [ # TLS or non-TLS to TLS or non-TLS: referrer origin is sent (yes, even for downgrades) ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/'), @@ -157,7 +157,7 @@ class MixinOrigin(object): ] -class MixinStrictOrigin(object): +class MixinStrictOrigin: scenarii = [ # TLS or non-TLS to TLS or non-TLS: referrer origin is sent but not for downgrades ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/'), @@ -176,7 +176,7 @@ class MixinStrictOrigin(object): ] -class MixinOriginWhenCrossOrigin(object): +class MixinOriginWhenCrossOrigin: scenarii = [ # Same origin (protocol, host, port): send referrer ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), @@ -211,7 +211,7 @@ class MixinOriginWhenCrossOrigin(object): ] -class MixinStrictOriginWhenCrossOrigin(object): +class MixinStrictOriginWhenCrossOrigin: scenarii = [ # Same origin (protocol, host, port): send referrer ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), @@ -255,7 +255,7 @@ class MixinStrictOriginWhenCrossOrigin(object): ] -class MixinUnsafeUrl(object): +class MixinUnsafeUrl: scenarii = [ # TLS to TLS: send referrer ('https://example.com/sekrit.html', 'http://not.example.com/', b'https://example.com/sekrit.html'), diff --git a/tests/test_squeues.py b/tests/test_squeues.py index f0f3dd4c6..5ad8035f7 100644 --- a/tests/test_squeues.py +++ b/tests/test_squeues.py @@ -36,7 +36,7 @@ def nonserializable_object_test(self): self.assertRaises(ValueError, q.push, lambda x: x) else: # Use a different unpickleable object - class A(object): + class A: pass a = A() diff --git a/tests/test_utils_deprecate.py b/tests/test_utils_deprecate.py index b3a90d314..b17e17f2f 100644 --- a/tests/test_utils_deprecate.py +++ b/tests/test_utils_deprecate.py @@ -11,7 +11,7 @@ class MyWarning(UserWarning): pass -class SomeBaseClass(object): +class SomeBaseClass: pass @@ -155,7 +155,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): class OutdatedUserClass1a(DeprecatedName): pass - class UnrelatedClass(object): + class UnrelatedClass: pass class OldStyleClass: @@ -191,7 +191,7 @@ class WarnWhenSubclassedTest(unittest.TestCase): class OutdatedUserClass2a(DeprecatedName): pass - class UnrelatedClass(object): + class UnrelatedClass: pass class OldStyleClass: diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index ec5b4c596..8cb8df15b 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -73,7 +73,7 @@ class ToBytesTest(unittest.TestCase): class MemoizedMethodTest(unittest.TestCase): def test_memoizemethod_noargs(self): - class A(object): + class A: @memoizemethod_noargs def cached(self): @@ -153,7 +153,7 @@ class UtilsPythonTestCase(unittest.TestCase): self.assertFalse(equal_attributes(a, b, [compare_z, 'x'])) def test_weakkeycache(self): - class _Weakme(object): + class _Weakme: pass _values = count() @@ -176,14 +176,14 @@ class UtilsPythonTestCase(unittest.TestCase): def f2(a, b=None, c=None): pass - class A(object): + class A: def __init__(self, a, b, c): pass def method(self, a, b, c): pass - class Callable(object): + class Callable: def __call__(self, a, b, c): pass diff --git a/tests/test_utils_reqser.py b/tests/test_utils_reqser.py index 06d9c004c..c7572f02c 100644 --- a/tests/test_utils_reqser.py +++ b/tests/test_utils_reqser.py @@ -126,7 +126,7 @@ class RequestSerializationTest(unittest.TestCase): self.assertRaises(ValueError, request_to_dict, r) -class TestSpiderMixin(object): +class TestSpiderMixin: def __mixin_callback(self, response): pass From 533131a30fa944688dc54dd82a581739d1ed247c Mon Sep 17 00:00:00 2001 From: sakshamb2113 <44064539+sakshamb2113@users.noreply.github.com> Date: Tue, 17 Mar 2020 14:42:49 +0530 Subject: [PATCH 264/313] Remove Guppy-specific code and documentation (#4343) --- docs/topics/leaks.rst | 64 +++---------------------------------- scrapy/extensions/telnet.py | 6 ---- 2 files changed, 5 insertions(+), 65 deletions(-) diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 4ee447065..ceb708c7e 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -17,8 +17,8 @@ what is known as a "memory leak". To help debugging memory leaks, Scrapy provides a built-in mechanism for tracking objects references called :ref:`trackref <topics-leaks-trackrefs>`, -and you can also use a third-party library called :ref:`Guppy -<topics-leaks-guppy>` for more advanced memory debugging (see below for more +and you can also use a third-party library called :ref:`muppy +<topics-leaks-muppy>` for more advanced memory debugging (see below for more info). Both mechanisms must be used from the :ref:`Telnet Console <topics-telnetconsole>`. @@ -193,9 +193,9 @@ Here are the functions available in the :mod:`~scrapy.utils.trackref` module. ``None`` if none is found. Use :func:`print_live_refs` first to get a list of all tracked live objects per class name. -.. _topics-leaks-guppy: +.. _topics-leaks-muppy: -Debugging memory leaks with Guppy +Debugging memory leaks with muppy ================================= ``trackref`` provides a very convenient mechanism for tracking down memory @@ -203,63 +203,9 @@ leaks, but it only keeps track of the objects that are more likely to cause memory leaks (Requests, Responses, Items, and Selectors). However, there are other cases where the memory leaks could come from other (more or less obscure) objects. If this is your case, and you can't find your leaks using ``trackref``, -you still have another resource: the `Guppy library`_. -If you're using Python3, see :ref:`topics-leaks-muppy`. +you still have another resource: the muppy library. -.. _Guppy library: https://pypi.org/project/guppy/ -If you use ``pip``, you can install Guppy with the following command:: - - pip install guppy - -The telnet console also comes with a built-in shortcut (``hpy``) for accessing -Guppy heap objects. Here's an example to view all Python objects available in -the heap using Guppy: - ->>> x = hpy.heap() ->>> x.bytype -Partition of a set of 297033 objects. Total size = 52587824 bytes. - Index Count % Size % Cumulative % Type - 0 22307 8 16423880 31 16423880 31 dict - 1 122285 41 12441544 24 28865424 55 str - 2 68346 23 5966696 11 34832120 66 tuple - 3 227 0 5836528 11 40668648 77 unicode - 4 2461 1 2222272 4 42890920 82 type - 5 16870 6 2024400 4 44915320 85 function - 6 13949 5 1673880 3 46589200 89 types.CodeType - 7 13422 5 1653104 3 48242304 92 list - 8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern - 9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers -<1676 more rows. Type e.g. '_.more' to view.> - -You can see that most space is used by dicts. Then, if you want to see from -which attribute those dicts are referenced, you could do: - ->>> x.bytype[0].byvia -Partition of a set of 22307 objects. Total size = 16423880 bytes. - Index Count % Size % Cumulative % Referred Via: - 0 10982 49 9416336 57 9416336 57 '.__dict__' - 1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals' - 2 3097 14 1122904 7 13220744 80 - 3 990 4 277200 2 13497944 82 "['cookies']" - 4 987 4 276360 2 13774304 84 "['cache']" - 5 985 4 275800 2 14050104 86 "['meta']" - 6 897 4 251160 2 14301264 87 '[2]' - 7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']" - 8 672 3 188160 1 14686312 89 "['cb_kwargs']" - 9 27 0 155016 1 14841328 90 '[1]' -<333 more rows. Type e.g. '_.more' to view.> - -As you can see, the Guppy module is very powerful but also requires some deep -knowledge about Python internals. For more info about Guppy, refer to the -`Guppy documentation`_. - -.. _Guppy documentation: http://guppy-pe.sourceforge.net/ - -.. _topics-leaks-muppy: - -Debugging memory leaks with muppy -================================= You can use muppy from `Pympler`_. .. _Pympler: https://pypi.org/project/Pympler/ diff --git a/scrapy/extensions/telnet.py b/scrapy/extensions/telnet.py index 26b214ee2..04ffd7235 100644 --- a/scrapy/extensions/telnet.py +++ b/scrapy/extensions/telnet.py @@ -26,11 +26,6 @@ from scrapy.utils.engine import print_engine_status from scrapy.utils.reactor import listen_tcp from scrapy.utils.decorators import defers -try: - import guppy - hpy = guppy.hpy() -except ImportError: - hpy = None logger = logging.getLogger(__name__) @@ -110,7 +105,6 @@ class TelnetConsole(protocol.ServerFactory): 'est': lambda: print_engine_status(self.crawler.engine), 'p': pprint.pprint, 'prefs': print_live_refs, - 'hpy': hpy, 'help': "This is Scrapy telnet console. For more info see: " "https://docs.scrapy.org/en/latest/topics/telnetconsole.html", } From 9ab45325ff6746f5c5325940fbbff1e020dd999a Mon Sep 17 00:00:00 2001 From: "Matsievskiy S.V" <matsievskiysv@gmail.com> Date: Tue, 17 Mar 2020 18:45:00 +0300 Subject: [PATCH 265/313] edit zsh completion - Fix bug introduced in https://github.com/scrapy/scrapy/pull/4291 - Enforce `[command] [options] [arguments]` syntax. Do not allow options after arguments - Exclude already used option aliases from completion list --- extras/scrapy_zsh_completion | 76 ++++++++++++++++++------------------ 1 file changed, 38 insertions(+), 38 deletions(-) diff --git a/extras/scrapy_zsh_completion b/extras/scrapy_zsh_completion index 33f46eda8..e2f2dc82b 100644 --- a/extras/scrapy_zsh_completion +++ b/extras/scrapy_zsh_completion @@ -14,40 +14,40 @@ _scrapy() { ;; args) case $words[1] in - bench) + (bench) _scrapy_glb_opts ;; - fetch) + (fetch) local options=( '--headers[print response HTTP headers instead of body]' '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' - '--spider[use this spider]:spider:_scrapy_spiders' + '--spider=[use this spider]:spider:_scrapy_spiders' '1::URL:_httpie_urls' ) _scrapy_glb_opts $options ;; - genspider) + (genspider) local options=( - {-l,--list}'[List available templates]' - {-e,--edit}'[Edit spider after creating it]' + {'(--list)-l','(-l)--list'}'[List available templates]' + {'(--edit)-e','(-e)--edit'}'[Edit spider after creating it]' '--force[If the spider already exists, overwrite it with the template]' - {-d,--dump=}'[Dump template to standard output]:template:(basic crawl csvfeed xmlfeed)' - {-t,--template=}'[Uses a custom template]:template:(basic crawl csvfeed xmlfeed)' + {'(--dump)-d','(-d)--dump='}'[Dump template to standard output]:template:(basic crawl csvfeed xmlfeed)' + {'(--template)-t','(-t)--template='}'[Uses a custom template]:template:(basic crawl csvfeed xmlfeed)' '1:name:(NAME)' '2:domain:_httpie_urls' ) _scrapy_glb_opts $options ;; - runspider) + (runspider) local options=( - {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files' - {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)' + {'(--output)-o','(-o)--output='}'[dump scraped items into FILE (use - for stdout)]:file:_files' + {'(--output-format)-t','(-t)--output-format='}'[format to use for dumping items with -o]:format:(FORMAT)' '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' '1:spider file:_files -g \*.py' ) _scrapy_glb_opts $options ;; - settings) + (settings) local options=( '--get=[print raw setting value]:option:(SETTING)' '--getbool=[print setting value, interpreted as a boolean]:option:(SETTING)' @@ -57,77 +57,77 @@ _scrapy() { ) _scrapy_glb_opts $options ;; - shell) + (shell) local options=( '-c[evaluate the code in the shell, print the result and exit]:code:(CODE)' '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' - '--spider[use this spider]:spider:_scrapy_spiders' + '--spider=[use this spider]:spider:_scrapy_spiders' '::file:_files -g \*.html' '::URL:_httpie_urls' ) _scrapy_glb_opts $options ;; - startproject) + (startproject) local options=( '1:name:(NAME)' '2:dir:_dir_list' ) _scrapy_glb_opts $options ;; - version) + (version) local options=( - {-v,--verbose}'[also display twisted/python/platform info (useful for bug reports)]' + {'(--verbose)-v','(-v)--verbose'}'[also display twisted/python/platform info (useful for bug reports)]' ) _scrapy_glb_opts $options ;; - view) + (view) local options=( '--no-redirect[do not handle HTTP 3xx status codes and print response as-is]' - '--spider[use this spider]:spider:_scrapy_spiders' + '--spider=[use this spider]:spider:_scrapy_spiders' '1:URL:_httpie_urls' ) _scrapy_glb_opts $options ;; - check) + (check) local options=( - '(- 1 *)'{-l,--list}'[only list contracts, without checking them]' - {-v,--verbose}'[print contract tests for all spiders]' + {'(--list)-l','(-l)--list'}'[only list contracts, without checking them]' + {'(--verbose)-v','(-v)--verbose'}'[print contract tests for all spiders]' '1:spider:_scrapy_spiders' ) _scrapy_glb_opts $options ;; - crawl) + (crawl) local options=( - {-o,--output}'[dump scraped items into FILE (use - for stdout)]:file:_files' - {-t,--output-format}'[format to use for dumping items with -o]:format:(FORMAT)' + {'(--output)-o','(-o)--output='}'[dump scraped items into FILE (use - for stdout)]:file:_files' + {'(--output-format)-t','(-t)--output-format='}'[format to use for dumping items with -o]:format:(FORMAT)' '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' '1:spider:_scrapy_spiders' ) _scrapy_glb_opts $options ;; - edit) + (edit) local options=( - '1:spider:_scrapy_spiders' + '1:spider:_scrapy_spiders' ) _scrapy_glb_opts $options ;; - list) + (list) _scrapy_glb_opts ;; - parse) + (parse) local options=( '*-a[set spider argument (may be repeated)]:value pair:(NAME=VALUE)' - '--spider[use this spider without looking for one]:spider:_scrapy_spiders' + '--spider=[use this spider without looking for one]:spider:_scrapy_spiders' '--pipelines[process items through pipelines]' "--nolinks[don't show links to follow (extracted requests)]" "--noitems[don't show scraped items]" '--nocolour[avoid using pygments to colorize the output]' - {-r,--rules}'[use CrawlSpider rules to discover the callback]' - {-c,--callback=}'[use this callback for parsing, instead looking for a callback]:callback:(CALLBACK)' - {-m,--meta=}'[inject extra meta into the Request, it must be a valid raw json string]:meta:(META)' + {'(--rules)-r','(-r)--rules'}'[use CrawlSpider rules to discover the callback]' + {'(--callback)-c','(-c)--callback'}'[use this callback for parsing, instead looking for a callback]:callback:(CALLBACK)' + {'(--meta)-m','(-m)--meta='}'[inject extra meta into the Request, it must be a valid raw json string]:meta:(META)' '--cbkwargs=[inject extra callback kwargs into the Request, it must be a valid raw json string]:arguments:(CBKWARGS)' - {-d,--depth=}'[maximum depth for parsing requests (default: 1)]:depth:(DEPTH)' - {-v,--verbose}'[print each depth level one by one]' + {'(--depth)-d','(-d)--depth='}'[maximum depth for parsing requests (default: 1)]:depth:(DEPTH)' + {'(--verbose)-v','(-v)--verbose'}'[print each depth level one by one]' '1:URL:_httpie_urls' ) _scrapy_glb_opts $options @@ -162,7 +162,7 @@ _scrapy_cmds() { if [[ $(scrapy -h | grep -s "no active project") == "" ]]; then commands=(${commands[@]} ${project_commands[@]}) fi - _describe -t common-commands 'common commands' commands + _describe -t common-commands 'common commands' commands && ret=0 } _scrapy_glb_opts() { @@ -172,13 +172,13 @@ _scrapy_glb_opts() { '(--nolog)--logfile=[log file. if omitted stderr will be used]:file:_files' '--pidfile=[write process ID to FILE]:file:_files' '--profile=[write python cProfile stats to FILE]:file:_files' - '(--nolog)'{-L,--loglevel=}'[log level (default: INFO)]:log level:(DEBUG INFO WARN ERROR)' + {'(--loglevel --nolog)-L','(-L --nolog)--loglevel='}'[log level (default: INFO)]:log level:(DEBUG INFO WARN ERROR)' '(-L --loglevel --logfile)--nolog[disable logging completely]' '--pdb[enable pdb on failure]' '*'{-s,--set=}'[set/override setting (may be repeated)]:value pair:(NAME=VALUE)' ) options=(${options[@]} "$@") - _arguments $options + _arguments -A "-*" $options && ret=0 } _httpie_urls() { From 6c747953f97d60e5668cff1af3f091979ceaaa83 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= <adrian@chaves.io> Date: Wed, 18 Mar 2020 18:33:41 +0100 Subject: [PATCH 266/313] Cover 2.0.1 in the release notes (#4437) --- docs/news.rst | 16 ++++++++++++++++ 1 file changed, 16 insertions(+) diff --git a/docs/news.rst b/docs/news.rst index dd5e00223..e9b7140cd 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -3,6 +3,22 @@ Release notes ============= +.. _release-2.0.1: + +Scrapy 2.0.1 (2020-03-18) +------------------------- + +* :meth:`Response.follow_all <scrapy.http.Response.follow_all>` now supports + an empty URL iterable as input (:issue:`4408`, :issue:`4420`) + +* Removed top-level :mod:`~twisted.internet.reactor` imports to prevent + errors about the wrong Twisted reactor being installed when setting a + different Twisted reactor using :setting:`TWISTED_REACTOR` (:issue:`4401`, + :issue:`4406`) + +* Fixed tests (:issue:`4422`) + + .. _release-2.0.0: Scrapy 2.0.0 (2020-03-03) From ca08e04198b94bd9583704f86316b57af3408adc Mon Sep 17 00:00:00 2001 From: Aditya <k.aditya00@gmail.com> Date: Fri, 20 Mar 2020 02:31:35 +0530 Subject: [PATCH 267/313] [docs] update redirect links python2 -> python3 --- docs/topics/downloader-middleware.rst | 5 ++--- docs/topics/email.rst | 2 +- docs/topics/exporters.rst | 8 ++++---- docs/topics/extensions.rst | 2 +- docs/topics/items.rst | 6 +++--- docs/topics/logging.rst | 16 ++++++++-------- docs/topics/request-response.rst | 10 +++++----- docs/topics/selectors.rst | 2 +- docs/topics/settings.rst | 6 +++--- docs/topics/spider-middleware.rst | 6 +++--- 10 files changed, 31 insertions(+), 32 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 73648994d..61a3806fb 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -739,7 +739,7 @@ HttpProxyMiddleware This middleware sets the HTTP proxy to use for requests, by setting the ``proxy`` meta value for :class:`~scrapy.http.Request` objects. - Like the Python standard library modules `urllib`_ and `urllib2`_, it obeys + Like the Python standard library module `urllib.request`_, it obeys the following environment variables: * ``http_proxy`` @@ -751,8 +751,7 @@ HttpProxyMiddleware Keep in mind this value will take precedence over ``http_proxy``/``https_proxy`` environment variables, and it will also ignore ``no_proxy`` environment variable. -.. _urllib: https://docs.python.org/2/library/urllib.html -.. _urllib2: https://docs.python.org/2/library/urllib2.html +.. _urllib.request: https://docs.python.org/3/library/urllib.request.html RedirectMiddleware ------------------ diff --git a/docs/topics/email.rst b/docs/topics/email.rst index 72bf52227..aed3deb2e 100644 --- a/docs/topics/email.rst +++ b/docs/topics/email.rst @@ -15,7 +15,7 @@ IO of the crawler. It also provides a simple API for sending attachments and it's very easy to configure, with a few :ref:`settings <topics-email-settings>`. -.. _smtplib: https://docs.python.org/2/library/smtplib.html +.. _smtplib: https://docs.python.org/3/library/smtplib.html Quick example ============= diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index e52682690..4ba8714bd 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -320,7 +320,7 @@ CsvItemExporter Color TV,1200 DVD player,200 -.. _csv.writer: https://docs.python.org/2/library/csv.html#csv.writer +.. _csv.writer: https://docs.python.org/3/library/csv.html#csv.writer PickleItemExporter ------------------ @@ -342,7 +342,7 @@ PickleItemExporter Pickle isn't a human readable format, so no output examples are provided. -.. _pickle module documentation: https://docs.python.org/2/library/pickle.html +.. _pickle module documentation: https://docs.python.org/3/library/pickle.html PprintItemExporter ------------------ @@ -393,7 +393,7 @@ JsonItemExporter stream-friendly format, consider using :class:`JsonLinesItemExporter` instead, or splitting the output in multiple chunks. -.. _JSONEncoder: https://docs.python.org/2/library/json.html#json.JSONEncoder +.. _JSONEncoder: https://docs.python.org/3/library/json.html#json.JSONEncoder JsonLinesItemExporter --------------------- @@ -417,7 +417,7 @@ JsonLinesItemExporter Unlike the one produced by :class:`JsonItemExporter`, the format produced by this exporter is well suited for serializing large amounts of data. -.. _JSONEncoder: https://docs.python.org/2/library/json.html#json.JSONEncoder +.. _JSONEncoder: https://docs.python.org/3/library/json.html#json.JSONEncoder MarshalItemExporter ------------------- diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index 94fd2e36e..f57e37e6f 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -372,5 +372,5 @@ For more info see `Debugging in Python`_. This extension only works on POSIX-compliant platforms (i.e. not Windows). -.. _Python debugger: https://docs.python.org/2/library/pdb.html +.. _Python debugger: https://docs.python.org/3/library/pdb.html .. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/ diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 44643cb67..36731571e 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -24,7 +24,7 @@ serialization can be customized using Item fields metadata, :mod:`trackref` tracks Item instances to help find memory leaks (see :ref:`topics-leaks-trackrefs`), etc. -.. _dictionary-like: https://docs.python.org/2/library/stdtypes.html#dict +.. _dictionary-like: https://docs.python.org/3/library/stdtypes.html#dict .. _topics-items-declaring: @@ -249,7 +249,7 @@ Item objects :class:`Field` objects used in the :ref:`Item declaration <topics-items-declaring>`. -.. _dict API: https://docs.python.org/2/library/stdtypes.html#dict +.. _dict API: https://docs.python.org/3/library/stdtypes.html#dict Field objects ============= @@ -262,7 +262,7 @@ Field objects to support the :ref:`item declaration syntax <topics-items-declaring>` based on class attributes. -.. _dict: https://docs.python.org/2/library/stdtypes.html#dict +.. _dict: https://docs.python.org/3/library/stdtypes.html#dict Other classes related to Item diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index d4d22d889..a85e1a769 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -83,10 +83,10 @@ path:: .. seealso:: - Module logging, `HowTo <https://docs.python.org/2/howto/logging.html>`_ + Module logging, `HowTo <https://docs.python.org/3/howto/logging.html>`_ Basic Logging Tutorial - Module logging, `Loggers <https://docs.python.org/2/library/logging.html#logger-objects>`_ + Module logging, `Loggers <https://docs.python.org/3/library/logging.html#logger-objects>`_ Further documentation on loggers .. _topics-logging-from-spiders: @@ -166,13 +166,13 @@ possible levels listed in :ref:`topics-logging-levels`. :setting:`LOG_FORMAT` and :setting:`LOG_DATEFORMAT` specify formatting strings used as layouts for all messages. Those strings can contain any placeholders listed in `logging's logrecord attributes docs -<https://docs.python.org/2/library/logging.html#logrecord-attributes>`_ and +<https://docs.python.org/3/library/logging.html#logrecord-attributes>`_ and `datetime's strftime and strptime directives -<https://docs.python.org/2/library/datetime.html#strftime-and-strptime-behavior>`_ +<https://docs.python.org/3/library/datetime.html#strftime-and-strptime-behavior>`_ respectively. If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the Scrapy -component that prints the log. It is unset by default, hence logs contain the +component that prints the log. It is unset by default, hence logs contain the Scrapy component responsible for that log output. Command-line options @@ -190,7 +190,7 @@ to override some of the Scrapy settings regarding logging. .. seealso:: - Module `logging.handlers <https://docs.python.org/2/library/logging.handlers.html>`_ + Module `logging.handlers <https://docs.python.org/3/library/logging.handlers.html>`_ Further documentation on available handlers .. _custom-log-formats: @@ -201,7 +201,7 @@ Custom Log Formats A custom log format can be set for different actions by extending :class:`~scrapy.logformatter.LogFormatter` class and making :setting:`LOG_FORMATTER` point to your new class. - + .. autoclass:: scrapy.logformatter.LogFormatter :members: @@ -276,6 +276,6 @@ scrapy.utils.log module Refer to :ref:`run-from-script` for more details about using Scrapy this way. -.. _logging.basicConfig(): https://docs.python.org/2/library/logging.html#logging.basicConfig +.. _logging.basicConfig(): https://docs.python.org/3/library/logging.html#logging.basicConfig diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index b2a60ff39..6c5a08409 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -189,7 +189,7 @@ Request objects ``copy()`` or ``replace()`` methods, and can also be accessed, in your spider, from the ``response.cb_kwargs`` attribute. - .. _shallow copied: https://docs.python.org/2/library/copy.html + .. _shallow copied: https://docs.python.org/3/library/copy.html .. method:: Request.copy() @@ -706,7 +706,7 @@ Response objects A :class:`twisted.internet.ssl.Certificate` object representing the server's SSL certificate. - + Only populated for ``https`` responses, ``None`` otherwise. .. method:: Response.copy() @@ -724,17 +724,17 @@ Response objects Constructs an absolute url by combining the Response's :attr:`url` with a possible relative url. - This is a wrapper over `urlparse.urljoin`_, it's merely an alias for + This is a wrapper over `urllib.parse.urljoin`_, it's merely an alias for making this call:: - urlparse.urljoin(response.url, url) + urllib.parse.urljoin(response.url, url) .. automethod:: Response.follow .. automethod:: Response.follow_all -.. _urlparse.urljoin: https://docs.python.org/2/library/urlparse.html#urlparse.urljoin +.. _urllib.parse.urljoin: https://docs.python.org/3/library/urllib.parse.html#urllib.parse.urljoin .. _topics-request-response-ref-response-subclasses: diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 1f7802c98..0f90b28c0 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -36,7 +36,7 @@ defines selectors to associate those styles with specific HTML elements. .. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/ .. _lxml: https://lxml.de/ -.. _ElementTree: https://docs.python.org/2/library/xml.etree.elementtree.html +.. _ElementTree: https://docs.python.org/3/library/xml.etree.elementtree.html .. _XPath: https://www.w3.org/TR/xpath/all/ .. _CSS: https://www.w3.org/TR/selectors .. _parsel: https://parsel.readthedocs.io/en/latest/ diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index dc6843d75..d78a6253e 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -28,7 +28,7 @@ The value of ``SCRAPY_SETTINGS_MODULE`` should be in Python path syntax, e.g. ``myproject.settings``. Note that the settings module should be on the Python `import search path`_. -.. _import search path: https://docs.python.org/2/tutorial/modules.html#the-module-search-path +.. _import search path: https://docs.python.org/3/tutorial/modules.html#the-module-search-path .. _populating-settings: @@ -902,7 +902,7 @@ Default: ``'%(asctime)s [%(name)s] %(levelname)s: %(message)s'`` String for formatting log messages. Refer to the `Python logging documentation`_ for the whole list of available placeholders. -.. _Python logging documentation: https://docs.python.org/2/library/logging.html#logrecord-attributes +.. _Python logging documentation: https://docs.python.org/3/library/logging.html#logrecord-attributes .. setting:: LOG_DATEFORMAT @@ -915,7 +915,7 @@ String for formatting date/time, expansion of the ``%(asctime)s`` placeholder in :setting:`LOG_FORMAT`. Refer to the `Python datetime documentation`_ for the whole list of available directives. -.. _Python datetime documentation: https://docs.python.org/2/library/datetime.html#strftime-and-strptime-behavior +.. _Python datetime documentation: https://docs.python.org/3/library/datetime.html#strftime-and-strptime-behavior .. setting:: LOG_FORMATTER diff --git a/docs/topics/spider-middleware.rst b/docs/topics/spider-middleware.rst index 0e8210130..3d7450c86 100644 --- a/docs/topics/spider-middleware.rst +++ b/docs/topics/spider-middleware.rst @@ -173,18 +173,18 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`. :type spider: :class:`~scrapy.spiders.Spider` object .. method:: from_crawler(cls, crawler) - + If present, this classmethod is called to create a middleware instance from a :class:`~scrapy.crawler.Crawler`. It must return a new instance of the middleware. Crawler object provides access to all Scrapy core components like settings and signals; it is a way for middleware to access them and hook its functionality into Scrapy. - + :param crawler: crawler that uses this middleware :type crawler: :class:`~scrapy.crawler.Crawler` object -.. _Exception: https://docs.python.org/2/library/exceptions.html#exceptions.Exception +.. _Exception: https://docs.python.org/3/library/exceptions.html#Exception .. _topics-spider-middleware-ref: From f37b1bdc5616f67460c645e26c49f9d5b34e3631 Mon Sep 17 00:00:00 2001 From: Aditya <k.aditya00@gmail.com> Date: Fri, 20 Mar 2020 05:22:51 +0530 Subject: [PATCH 268/313] [docs] update redirect links to python3 --- docs/intro/tutorial.rst | 10 +++++----- docs/topics/contracts.rst | 4 +--- docs/topics/downloader-middleware.rst | 11 +++-------- docs/topics/dynamic-content.rst | 10 ++++------ docs/topics/email.rst | 4 +--- docs/topics/exporters.rst | 20 ++++++-------------- docs/topics/extensions.rst | 3 +-- docs/topics/items.rst | 21 ++++++--------------- docs/topics/logging.rst | 15 +++++---------- docs/topics/request-response.rst | 8 ++------ docs/topics/selectors.rst | 3 +-- docs/topics/spider-middleware.rst | 6 +----- docs/topics/spiders.rst | 4 +--- docs/topics/telnetconsole.rst | 11 ++++------- scrapy/item.py | 4 +--- 15 files changed, 42 insertions(+), 92 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index 1768badbb..ab6fd4829 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -25,16 +25,16 @@ Scrapy. If you're already familiar with other languages, and want to learn Python quickly, the `Python Tutorial`_ is a good resource. If you're new to programming and want to start with Python, the following books -may be useful to you: +may be useful to you: * `Automate the Boring Stuff With Python`_ -* `How To Think Like a Computer Scientist`_ +* `How To Think Like a Computer Scientist`_ -* `Learn Python 3 The Hard Way`_ +* `Learn Python 3 The Hard Way`_ You can also take a look at `this list of Python resources for non-programmers`_, -as well as the `suggested resources in the learnpython-subreddit`_. +as well as the `suggested resources in the learnpython-subreddit`_. .. _Python: https://www.python.org/ .. _this list of Python resources for non-programmers: https://wiki.python.org/moin/BeginnersGuide/NonProgrammers @@ -62,7 +62,7 @@ This will create a ``tutorial`` directory with the following contents:: __init__.py items.py # project items definition file - + middlewares.py # project middlewares file pipelines.py # project pipelines file diff --git a/docs/topics/contracts.rst b/docs/topics/contracts.rst index 43db8f101..319f577bc 100644 --- a/docs/topics/contracts.rst +++ b/docs/topics/contracts.rst @@ -136,7 +136,7 @@ Detecting check runs ==================== When ``scrapy check`` is running, the ``SCRAPY_CHECK`` environment variable is -set to the ``true`` string. You can use `os.environ`_ to perform any change to +set to the ``true`` string. You can use :data:`os.environ` to perform any change to your spiders or your settings when ``scrapy check`` is used:: import os @@ -148,5 +148,3 @@ your spiders or your settings when ``scrapy check`` is used:: def __init__(self): if os.environ.get('SCRAPY_CHECK'): pass # Do some scraper adjustments when a check is running - -.. _os.environ: https://docs.python.org/3/library/os.html#os.environ diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 61a3806fb..d7ec53bfa 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -739,7 +739,7 @@ HttpProxyMiddleware This middleware sets the HTTP proxy to use for requests, by setting the ``proxy`` meta value for :class:`~scrapy.http.Request` objects. - Like the Python standard library module `urllib.request`_, it obeys + Like the Python standard library module :mod:`urllib.request`, it obeys the following environment variables: * ``http_proxy`` @@ -751,8 +751,6 @@ HttpProxyMiddleware Keep in mind this value will take precedence over ``http_proxy``/``https_proxy`` environment variables, and it will also ignore ``no_proxy`` environment variable. -.. _urllib.request: https://docs.python.org/3/library/urllib.request.html - RedirectMiddleware ------------------ @@ -982,7 +980,7 @@ RobotsTxtMiddleware Scrapy ships with support for the following robots.txt_ parsers: * :ref:`Protego <protego-parser>` (default) - * :ref:`RobotFileParser <python-robotfileparser>` + * :class:`~urllib.robotparser.RobotFileParser` * :ref:`Reppy <reppy-parser>` * :ref:`Robotexclusionrulesparser <rerp-parser>` @@ -1030,13 +1028,10 @@ Based on `Protego <https://github.com/scrapy/protego>`_: Scrapy uses this parser by default. -.. _python-robotfileparser: - RobotFileParser ~~~~~~~~~~~~~~~ -Based on `RobotFileParser -<https://docs.python.org/3.7/library/urllib.robotparser.html>`_: +Based on :class:`~urllib.robotparser.RobotFileParser`: * is Python's built-in robots.txt_ parser diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index b98133676..22bcac268 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -115,7 +115,7 @@ data from it depends on the type of response: - If the response is HTML or XML, use :ref:`selectors <topics-selectors>` as usual. -- If the response is JSON, use `json.loads`_ to load the desired data from +- If the response is JSON, use :func:`json.loads` to load the desired data from :attr:`response.text <scrapy.http.TextResponse.text>`:: data = json.loads(response.text) @@ -130,7 +130,7 @@ data from it depends on the type of response: - If the response is JavaScript, or HTML with a ``<script/>`` element containing the desired data, see :ref:`topics-parsing-javascript`. -- If the response is CSS, use a `regular expression`_ to extract the desired +- If the response is CSS, use :mod:`re` to extract the desired data from :attr:`response.text <scrapy.http.TextResponse.text>`. .. _topics-parsing-images: @@ -168,8 +168,8 @@ JavaScript code: Once you have a string with the JavaScript code, you can extract the desired data from it: -- You might be able to use a `regular expression`_ to extract the desired - data in JSON format, which you can then parse with `json.loads`_. +- You might be able to use :mod:`re` to extract the desired + data in JSON format, which you can then parse with :func:`json.loads`. For example, if the JavaScript code contains a separate line like ``var data = {"field": "value"};`` you can extract that data as follows: @@ -241,9 +241,7 @@ along with `scrapy-selenium`_ for seamless integration. .. _headless browser: https://en.wikipedia.org/wiki/Headless_browser .. _JavaScript: https://en.wikipedia.org/wiki/JavaScript .. _js2xml: https://github.com/scrapinghub/js2xml -.. _json.loads: https://docs.python.org/3/library/json.html#json.loads .. _pytesseract: https://github.com/madmaze/pytesseract -.. _regular expression: https://docs.python.org/3/library/re.html .. _scrapy-selenium: https://github.com/clemfromspace/scrapy-selenium .. _scrapy-splash: https://github.com/scrapy-plugins/scrapy-splash .. _Selenium: https://www.selenium.dev/ diff --git a/docs/topics/email.rst b/docs/topics/email.rst index aed3deb2e..e347c3a35 100644 --- a/docs/topics/email.rst +++ b/docs/topics/email.rst @@ -7,7 +7,7 @@ Sending e-mail .. module:: scrapy.mail :synopsis: Email sending facility -Although Python makes sending e-mails relatively easy via the `smtplib`_ +Although Python makes sending e-mails relatively easy via the :mod:`smtplib` library, Scrapy provides its own facility for sending e-mails which is very easy to use and it's implemented using :doc:`Twisted non-blocking IO <twisted:core/howto/defer-intro>`, to avoid interfering with the non-blocking @@ -15,8 +15,6 @@ IO of the crawler. It also provides a simple API for sending attachments and it's very easy to configure, with a few :ref:`settings <topics-email-settings>`. -.. _smtplib: https://docs.python.org/3/library/smtplib.html - Quick example ============= diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index 4ba8714bd..f73c6728d 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -311,7 +311,7 @@ CsvItemExporter The additional keyword arguments of this ``__init__`` method are passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the - `csv.writer`_ ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method + :func:`csv.writer` ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method argument to customize this exporter. A typical output of this exporter would be:: @@ -320,8 +320,6 @@ CsvItemExporter Color TV,1200 DVD player,200 -.. _csv.writer: https://docs.python.org/3/library/csv.html#csv.writer - PickleItemExporter ------------------ @@ -335,15 +333,13 @@ PickleItemExporter :param protocol: The pickle protocol to use. :type protocol: int - For more information, refer to the `pickle module documentation`_. + For more information, refer :mod:`pickle`. The additional keyword arguments of this ``__init__`` method are passed to the :class:`BaseItemExporter` ``__init__`` method. Pickle isn't a human readable format, so no output examples are provided. -.. _pickle module documentation: https://docs.python.org/3/library/pickle.html - PprintItemExporter ------------------ @@ -372,8 +368,8 @@ JsonItemExporter Exports Items in JSON format to the specified file-like object, writing all objects as a list of objects. The additional ``__init__`` method arguments are passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover - arguments to the `JSONEncoder`_ ``__init__`` method, so you can use any - `JSONEncoder`_ ``__init__`` method argument to customize this exporter. + arguments to the :class:`~json.JSONEncoder` ``__init__`` method, so you can use any + :class:`~json.JSONEncoder` ``__init__`` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) @@ -393,8 +389,6 @@ JsonItemExporter stream-friendly format, consider using :class:`JsonLinesItemExporter` instead, or splitting the output in multiple chunks. -.. _JSONEncoder: https://docs.python.org/3/library/json.html#json.JSONEncoder - JsonLinesItemExporter --------------------- @@ -403,8 +397,8 @@ JsonLinesItemExporter Exports Items in JSON format to the specified file-like object, writing one JSON-encoded item per line. The additional ``__init__`` method arguments are passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to - the `JSONEncoder`_ ``__init__`` method, so you can use any `JSONEncoder`_ - ``__init__`` method argument to customize this exporter. + the :class:`~json.JSONEncoder` ``__init__`` method, so you can use any + :class:`~json.JSONEncoder` ``__init__`` method argument to customize this exporter. :param file: the file-like object to use for exporting the data. Its ``write`` method should accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc) @@ -417,8 +411,6 @@ JsonLinesItemExporter Unlike the one produced by :class:`JsonItemExporter`, the format produced by this exporter is well suited for serializing large amounts of data. -.. _JSONEncoder: https://docs.python.org/3/library/json.html#json.JSONEncoder - MarshalItemExporter ------------------- diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index f57e37e6f..1b8413abf 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -364,7 +364,7 @@ Debugger extension .. class:: Debugger -Invokes a `Python debugger`_ inside a running Scrapy process when a `SIGUSR2`_ +Invokes a :mod:`Python debugger <pdb>`: inside a running Scrapy process when a `SIGUSR2`_ signal is received. After the debugger is exited, the Scrapy process continues running normally. @@ -372,5 +372,4 @@ For more info see `Debugging in Python`_. This extension only works on POSIX-compliant platforms (i.e. not Windows). -.. _Python debugger: https://docs.python.org/3/library/pdb.html .. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/ diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 36731571e..2e5c88054 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -15,7 +15,7 @@ especially in a larger project with many spiders. To define common output data format Scrapy provides the :class:`Item` class. :class:`Item` objects are simple containers used to collect the scraped data. -They provide a `dictionary-like`_ API with a convenient syntax for declaring +They provide a :class:`dict` like API with a convenient syntax for declaring their available fields. Various Scrapy components use extra information provided by Items: @@ -24,8 +24,6 @@ serialization can be customized using Item fields metadata, :mod:`trackref` tracks Item instances to help find memory leaks (see :ref:`topics-leaks-trackrefs`), etc. -.. _dictionary-like: https://docs.python.org/3/library/stdtypes.html#dict - .. _topics-items-declaring: Declaring Items @@ -79,7 +77,7 @@ Working with Items Here are some examples of common tasks performed with items, using the ``Product`` item :ref:`declared above <topics-items-declaring>`. You will -notice the API is very similar to the `dict API`_. +notice the API is very similar to the :class:`dict` API. Creating items -------------- @@ -145,7 +143,7 @@ KeyError: 'Product does not support field: lala' Accessing all populated values ------------------------------ -To access all populated values, just use the typical `dict API`_: +To access all populated values, just use the typical :class:`dict`: >>> product.keys() ['price', 'name'] @@ -175,9 +173,7 @@ other item as well. If that is not the desired behavior, use a deep copy instead. -See the `documentation of the copy module`_ for more information. - -.. _documentation of the copy module: https://docs.python.org/3/library/copy.html +See :mod:`copy` for more information. To create a shallow copy of an item, you can either call :meth:`~scrapy.item.Item.copy` on an existing item @@ -235,7 +231,7 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard `dict API`_, including its ``__init__`` method, and + Items replicate the standard :class:`dict`, including its ``__init__`` method, and also provide the following additional API members: .. automethod:: copy @@ -249,22 +245,17 @@ Item objects :class:`Field` objects used in the :ref:`Item declaration <topics-items-declaring>`. -.. _dict API: https://docs.python.org/3/library/stdtypes.html#dict - Field objects ============= .. class:: Field([arg]) - The :class:`Field` class is just an alias to the built-in `dict`_ class and + The :class:`Field` class is just an alias to the built-in :class:`dict` class and doesn't provide any extra functionality or attributes. In other words, :class:`Field` objects are plain-old Python dicts. A separate class is used to support the :ref:`item declaration syntax <topics-items-declaring>` based on class attributes. -.. _dict: https://docs.python.org/3/library/stdtypes.html#dict - - Other classes related to Item ============================= diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index a85e1a769..df631b3dc 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -9,8 +9,7 @@ Logging explicit calls to the Python standard logging. Keep reading to learn more about the new logging system. -Scrapy uses `Python's builtin logging system -<https://docs.python.org/3/library/logging.html>`_ for event logging. We'll +Scrapy uses :mod:`logging` for event logging. We'll provide some simple examples to get you started, but for more advanced use-cases it's strongly suggested to read thoroughly its documentation. @@ -86,7 +85,7 @@ path:: Module logging, `HowTo <https://docs.python.org/3/howto/logging.html>`_ Basic Logging Tutorial - Module logging, `Loggers <https://docs.python.org/3/library/logging.html#logger-objects>`_ + Module logging, :class:`~logging.Logger` Further documentation on loggers .. _topics-logging-from-spiders: @@ -190,7 +189,7 @@ to override some of the Scrapy settings regarding logging. .. seealso:: - Module `logging.handlers <https://docs.python.org/3/library/logging.handlers.html>`_ + Module :mod:`logging.handlers` Further documentation on available handlers .. _custom-log-formats: @@ -256,10 +255,10 @@ scrapy.utils.log module In that case, its usage is not required but it's recommended. Another option when running custom scripts is to manually configure the logging. - To do this you can use `logging.basicConfig()`_ to set a basic root handler. + To do this you can use :func:`logging.basicConfig` to set a basic root handler. Note that :class:`~scrapy.crawler.CrawlerProcess` automatically calls ``configure_logging``, - so it is recommended to only use `logging.basicConfig()`_ together with + so it is recommended to only use :func:`logging.basicConfig` together with :class:`~scrapy.crawler.CrawlerRunner`. This is an example on how to redirect ``INFO`` or higher messages to a file:: @@ -275,7 +274,3 @@ scrapy.utils.log module Refer to :ref:`run-from-script` for more details about using Scrapy this way. - -.. _logging.basicConfig(): https://docs.python.org/3/library/logging.html#logging.basicConfig - - diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 6c5a08409..7260141e9 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -566,12 +566,10 @@ dealing with JSON requests. set to ``'POST'`` automatically. :type data: JSON serializable object - :param dumps_kwargs: Parameters that will be passed to underlying `json.dumps`_ method which is used to serialize + :param dumps_kwargs: Parameters that will be passed to underlying :func:`json.dumps` method which is used to serialize data into JSON format. :type dumps_kwargs: dict -.. _json.dumps: https://docs.python.org/3/library/json.html#json.dumps - JsonRequest usage example ------------------------- @@ -724,7 +722,7 @@ Response objects Constructs an absolute url by combining the Response's :attr:`url` with a possible relative url. - This is a wrapper over `urllib.parse.urljoin`_, it's merely an alias for + This is a wrapper over :func:`~urllib.parse.urljoin`, it's merely an alias for making this call:: urllib.parse.urljoin(response.url, url) @@ -734,8 +732,6 @@ Response objects .. automethod:: Response.follow_all -.. _urllib.parse.urljoin: https://docs.python.org/3/library/urllib.parse.html#urllib.parse.urljoin - .. _topics-request-response-ref-response-subclasses: Response subclasses diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 0f90b28c0..bb46ea80f 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -14,7 +14,7 @@ achieve this, such as: drawback: it's slow. * `lxml`_ is an XML parsing library (which also parses HTML) with a pythonic - API based on `ElementTree`_. (lxml is not part of the Python standard + API based on :mod:`~xml.etree.ElementTree`. (lxml is not part of the Python standard library.) Scrapy comes with its own mechanism for extracting data. They're called @@ -36,7 +36,6 @@ defines selectors to associate those styles with specific HTML elements. .. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/ .. _lxml: https://lxml.de/ -.. _ElementTree: https://docs.python.org/3/library/xml.etree.elementtree.html .. _XPath: https://www.w3.org/TR/xpath/all/ .. _CSS: https://www.w3.org/TR/selectors .. _parsel: https://parsel.readthedocs.io/en/latest/ diff --git a/docs/topics/spider-middleware.rst b/docs/topics/spider-middleware.rst index 3d7450c86..d49a2209d 100644 --- a/docs/topics/spider-middleware.rst +++ b/docs/topics/spider-middleware.rst @@ -140,7 +140,7 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`. :type response: :class:`~scrapy.http.Response` object :param exception: the exception raised - :type exception: `Exception`_ object + :type exception: :exc:`Exception` object :param spider: the spider which raised the exception :type spider: :class:`~scrapy.spiders.Spider` object @@ -183,10 +183,6 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`. :param crawler: crawler that uses this middleware :type crawler: :class:`~scrapy.crawler.Crawler` object - -.. _Exception: https://docs.python.org/3/library/exceptions.html#Exception - - .. _topics-spider-middleware-ref: Built-in spider middleware reference diff --git a/docs/topics/spiders.rst b/docs/topics/spiders.rst index 89609db7d..231db6cea 100644 --- a/docs/topics/spiders.rst +++ b/docs/topics/spiders.rst @@ -298,9 +298,7 @@ Keep in mind that spider arguments are only strings. The spider will not do any parsing on its own. If you were to set the ``start_urls`` attribute from the command line, you would have to parse it on your own into a list -using something like -`ast.literal_eval <https://docs.python.org/3/library/ast.html#ast.literal_eval>`_ -or `json.loads <https://docs.python.org/3/library/json.html#json.loads>`_ +using something like :func:`ast.literal_eval` or :func:`json.loads` and then set it as an attribute. Otherwise, you would cause iteration over a ``start_urls`` string (a very common python pitfall) diff --git a/docs/topics/telnetconsole.rst b/docs/topics/telnetconsole.rst index 47d8d393c..9802a34a2 100644 --- a/docs/topics/telnetconsole.rst +++ b/docs/topics/telnetconsole.rst @@ -40,10 +40,10 @@ the console you need to type:: Connected to localhost. Escape character is '^]'. Username: - Password: + Password: >>> -By default Username is ``scrapy`` and Password is autogenerated. The +By default Username is ``scrapy`` and Password is autogenerated. The autogenerated Password can be seen on Scrapy logs like the example below:: 2018-10-16 14:35:21 [scrapy.extensions.telnet] INFO: Telnet Password: 16f92501e8a59326 @@ -63,7 +63,7 @@ Available variables in the telnet console ========================================= The telnet console is like a regular Python shell running inside the Scrapy -process, so you can do anything from it including importing new modules, etc. +process, so you can do anything from it including importing new modules, etc. However, the telnet console comes with some default variables defined for convenience: @@ -89,13 +89,11 @@ convenience: +----------------+-------------------------------------------------------------------+ | ``prefs`` | for memory debugging (see :ref:`topics-leaks`) | +----------------+-------------------------------------------------------------------+ -| ``p`` | a shortcut to the `pprint.pprint`_ function | +| ``p`` | a shortcut to the :func:`pprint.pprint` function | +----------------+-------------------------------------------------------------------+ | ``hpy`` | for memory debugging (see :ref:`topics-leaks`) | +----------------+-------------------------------------------------------------------+ -.. _pprint.pprint: https://docs.python.org/library/pprint.html#pprint.pprint - Telnet console usage examples ============================= @@ -208,4 +206,3 @@ Default: ``None`` The password used for the telnet console, default behaviour is to have it autogenerated - diff --git a/scrapy/item.py b/scrapy/item.py index 1d39b48b2..748368932 100644 --- a/scrapy/item.py +++ b/scrapy/item.py @@ -121,9 +121,7 @@ class DictItem(MutableMapping, BaseItem): return self.__class__(self) def deepcopy(self): - """Return a `deep copy`_ of this item. - - .. _deep copy: https://docs.python.org/library/copy.html#copy.deepcopy + """Return a :func:`~copy.deepcopy` of this item. """ return deepcopy(self) From 532cd1d93ed58f038346ca6b753462563c85b489 Mon Sep 17 00:00:00 2001 From: Aditya <k.aditya00@gmail.com> Date: Fri, 20 Mar 2020 17:36:49 +0530 Subject: [PATCH 269/313] [fix] zope interface 5.0.0 unsupported --- scrapy/crawler.py | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 77a13d0c1..20990ea41 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -4,7 +4,15 @@ import signal import warnings from twisted.internet import defer -from zope.interface.verify import DoesNotImplement, verifyClass +from zope.interface.exceptions import DoesNotImplement + +try: + # zope >= 5.0 only supports MultipleInvalid + from zope.interface.exceptions import MultipleInvalid +except ImportError: + MultipleInvalid = None + +from zope.interface.verify import verifyClass from scrapy import signals, Spider from scrapy.core.engine import ExecutionEngine @@ -124,9 +132,10 @@ class CrawlerRunner: """ Get SpiderLoader instance from settings """ cls_path = settings.get('SPIDER_LOADER_CLASS') loader_cls = load_object(cls_path) + excs = (DoesNotImplement, MultipleInvalid) if MultipleInvalid else DoesNotImplement try: verifyClass(ISpiderLoader, loader_cls) - except DoesNotImplement: + except excs: warnings.warn( 'SPIDER_LOADER_CLASS (previously named SPIDER_MANAGER_CLASS) does ' 'not fully implement scrapy.interfaces.ISpiderLoader interface. ' From 80c69d68addef99f8b6ea5d3ec894752a06e2a9c Mon Sep 17 00:00:00 2001 From: Aditya <k.aditya00@gmail.com> Date: Tue, 24 Mar 2020 05:52:07 +0530 Subject: [PATCH 270/313] [docs] refactor python docs links using intersphinx --- docs/intro/tutorial.rst | 5 ++--- docs/topics/coroutines.rst | 5 ++--- docs/topics/downloader-middleware.rst | 4 +++- docs/topics/dynamic-content.rst | 10 ++++++---- docs/topics/exporters.rst | 4 ++-- docs/topics/extensions.rst | 2 +- docs/topics/items.rst | 18 ++++++++---------- docs/topics/logging.rst | 10 ++++------ docs/topics/request-response.rst | 14 ++++++-------- docs/topics/settings.rst | 18 +++++++----------- 10 files changed, 41 insertions(+), 49 deletions(-) diff --git a/docs/intro/tutorial.rst b/docs/intro/tutorial.rst index ab6fd4829..5f35dc936 100644 --- a/docs/intro/tutorial.rst +++ b/docs/intro/tutorial.rst @@ -287,8 +287,8 @@ to be scraped, you can at least get **some** data. Besides the :meth:`~scrapy.selector.SelectorList.getall` and :meth:`~scrapy.selector.SelectorList.get` methods, you can also use -the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular -expressions`_: +the :meth:`~scrapy.selector.SelectorList.re` method to extract using +:doc:`regular expressions <library/re>`: >>> response.css('title::text').re(r'Quotes.*') ['Quotes to Scrape'] @@ -305,7 +305,6 @@ with a selector (see :ref:`topics-developer-tools`). `Selector Gadget`_ is also a nice tool to quickly find CSS selector for visually selected elements, which works in many browsers. -.. _regular expressions: https://docs.python.org/3/library/re.html .. _Selector Gadget: https://selectorgadget.com/ diff --git a/docs/topics/coroutines.rst b/docs/topics/coroutines.rst index 487cf4c6c..5f61d6796 100644 --- a/docs/topics/coroutines.rst +++ b/docs/topics/coroutines.rst @@ -76,8 +76,8 @@ becomes:: Coroutines may be used to call asynchronous code. This includes other coroutines, functions that return Deferreds and functions that return -`awaitable objects`_ such as :class:`~asyncio.Future`. This means you can use -many useful Python libraries providing such code:: +:term:`awaitable objects <awaitable>` such as :class:`~asyncio.Future`. +This means you can use many useful Python libraries providing such code:: class MySpider(Spider): # ... @@ -107,4 +107,3 @@ Common use cases for asynchronous code include: :ref:`the screenshot pipeline example<ScreenshotPipeline>`). .. _aio-libs: https://github.com/aio-libs -.. _awaitable objects: https://docs.python.org/3/glossary.html#term-awaitable diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index d7ec53bfa..d309bbc49 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -980,7 +980,7 @@ RobotsTxtMiddleware Scrapy ships with support for the following robots.txt_ parsers: * :ref:`Protego <protego-parser>` (default) - * :class:`~urllib.robotparser.RobotFileParser` + * :ref:`RobotFileParser <python-robotfileparser>` * :ref:`Reppy <reppy-parser>` * :ref:`Robotexclusionrulesparser <rerp-parser>` @@ -1028,6 +1028,8 @@ Based on `Protego <https://github.com/scrapy/protego>`_: Scrapy uses this parser by default. +.. _python-robotfileparser: + RobotFileParser ~~~~~~~~~~~~~~~ diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index 22bcac268..a3f0d6ebb 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -130,8 +130,9 @@ data from it depends on the type of response: - If the response is JavaScript, or HTML with a ``<script/>`` element containing the desired data, see :ref:`topics-parsing-javascript`. -- If the response is CSS, use :mod:`re` to extract the desired - data from :attr:`response.text <scrapy.http.TextResponse.text>`. +- If the response is CSS, use a :doc:`regular expression <library/re>` to + extract the desired data from + :attr:`response.text <scrapy.http.TextResponse.text>`. .. _topics-parsing-images: @@ -168,8 +169,9 @@ JavaScript code: Once you have a string with the JavaScript code, you can extract the desired data from it: -- You might be able to use :mod:`re` to extract the desired - data in JSON format, which you can then parse with :func:`json.loads`. +- You might be able to use a :doc:`regular expression <library/re>` to + extract the desired data in JSON format, which you can then parse with + :func:`json.loads`. For example, if the JavaScript code contains a separate line like ``var data = {"field": "value"};`` you can extract that data as follows: diff --git a/docs/topics/exporters.rst b/docs/topics/exporters.rst index f73c6728d..de8b51195 100644 --- a/docs/topics/exporters.rst +++ b/docs/topics/exporters.rst @@ -311,7 +311,7 @@ CsvItemExporter The additional keyword arguments of this ``__init__`` method are passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the - :func:`csv.writer` ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method + :func:`csv.writer` function, so you can use any :func:`csv.writer` function argument to customize this exporter. A typical output of this exporter would be:: @@ -333,7 +333,7 @@ PickleItemExporter :param protocol: The pickle protocol to use. :type protocol: int - For more information, refer :mod:`pickle`. + For more information, see :mod:`pickle`. The additional keyword arguments of this ``__init__`` method are passed to the :class:`BaseItemExporter` ``__init__`` method. diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index 1b8413abf..0fc83e645 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -364,7 +364,7 @@ Debugger extension .. class:: Debugger -Invokes a :mod:`Python debugger <pdb>`: inside a running Scrapy process when a `SIGUSR2`_ +Invokes a :doc:`Python debugger <library/pdb>` inside a running Scrapy process when a `SIGUSR2`_ signal is received. After the debugger is exited, the Scrapy process continues running normally. diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 2e5c88054..78612f524 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -15,8 +15,8 @@ especially in a larger project with many spiders. To define common output data format Scrapy provides the :class:`Item` class. :class:`Item` objects are simple containers used to collect the scraped data. -They provide a :class:`dict` like API with a convenient syntax for declaring -their available fields. +They provide an API similar to :class:`dict` API with a convenient syntax +for declaring their available fields. Various Scrapy components use extra information provided by Items: exporters look at declared fields to figure out columns to export, @@ -143,7 +143,7 @@ KeyError: 'Product does not support field: lala' Accessing all populated values ------------------------------ -To access all populated values, just use the typical :class:`dict`: +To access all populated values, just use the typical :class:`dict` API: >>> product.keys() ['price', 'name'] @@ -160,11 +160,9 @@ Copying items To copy an item, you must first decide whether you want a shallow copy or a deep copy. -If your item contains mutable_ values like lists or dictionaries, a shallow -copy will keep references to the same mutable values across all different -copies. - -.. _mutable: https://docs.python.org/3/glossary.html#term-mutable +If your item contains :term:`mutable` values like lists or dictionaries, +a shallow copy will keep references to the same mutable values across all +different copies. For example, if you have an item with a list of tags, and you create a shallow copy of that item, both the original item and the copy have the same list of @@ -231,8 +229,8 @@ Item objects Return a new Item optionally initialized from the given argument. - Items replicate the standard :class:`dict`, including its ``__init__`` method, and - also provide the following additional API members: + Items replicate the standard :class:`dict` API, including its ``__init__`` + method, and also provide the following additional API members: .. automethod:: copy diff --git a/docs/topics/logging.rst b/docs/topics/logging.rst index df631b3dc..675e65ef1 100644 --- a/docs/topics/logging.rst +++ b/docs/topics/logging.rst @@ -82,10 +82,10 @@ path:: .. seealso:: - Module logging, `HowTo <https://docs.python.org/3/howto/logging.html>`_ + Module logging, :doc:`HowTo <howto/logging>` Basic Logging Tutorial - Module logging, :class:`~logging.Logger` + Module logging, :ref:`Loggers <logger>` Further documentation on loggers .. _topics-logging-from-spiders: @@ -164,10 +164,8 @@ possible levels listed in :ref:`topics-logging-levels`. :setting:`LOG_FORMAT` and :setting:`LOG_DATEFORMAT` specify formatting strings used as layouts for all messages. Those strings can contain any placeholders -listed in `logging's logrecord attributes docs -<https://docs.python.org/3/library/logging.html#logrecord-attributes>`_ and -`datetime's strftime and strptime directives -<https://docs.python.org/3/library/datetime.html#strftime-and-strptime-behavior>`_ +listed in :ref:`logging's logrecord attributes docs <logrecord-attributes>` and +:ref:`datetime's strftime and strptime directives <strftime-strptime-behavior>` respectively. If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the Scrapy diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 7260141e9..69a51a17c 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -174,9 +174,9 @@ Request objects See :ref:`topics-request-meta` for a list of special meta keys recognized by Scrapy. - This dict is `shallow copied`_ when the request is cloned using the - ``copy()`` or ``replace()`` methods, and can also be accessed, in your - spider, from the ``response.meta`` attribute. + This dict is :mod:`shallow copied <copy>` when the request is + cloned using the ``copy()`` or ``replace()`` methods, and can also be + accessed, in your spider, from the ``response.meta`` attribute. .. attribute:: Request.cb_kwargs @@ -185,11 +185,9 @@ Request objects for new Requests, which means by default callbacks only get a :class:`Response` object as argument. - This dict is `shallow copied`_ when the request is cloned using the - ``copy()`` or ``replace()`` methods, and can also be accessed, in your - spider, from the ``response.cb_kwargs`` attribute. - - .. _shallow copied: https://docs.python.org/3/library/copy.html + This dict is :mod:`shallow copied <copy>` when the request is + cloned using the ``copy()`` or ``replace()`` methods, and can also be + accessed, in your spider, from the ``response.cb_kwargs`` attribute. .. method:: Request.copy() diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index d78a6253e..0049dbfca 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -26,9 +26,7 @@ do this by using an environment variable, ``SCRAPY_SETTINGS_MODULE``. The value of ``SCRAPY_SETTINGS_MODULE`` should be in Python path syntax, e.g. ``myproject.settings``. Note that the settings module should be on the -Python `import search path`_. - -.. _import search path: https://docs.python.org/3/tutorial/modules.html#the-module-search-path +Python :ref:`import search path <tut-searchpath>`. .. _populating-settings: @@ -899,10 +897,9 @@ LOG_FORMAT Default: ``'%(asctime)s [%(name)s] %(levelname)s: %(message)s'`` -String for formatting log messages. Refer to the `Python logging documentation`_ for the whole list of available -placeholders. - -.. _Python logging documentation: https://docs.python.org/3/library/logging.html#logrecord-attributes +String for formatting log messages. Refer to the +:ref:`Python logging documentation <logrecord-attributes>` for the qwhole +list of available placeholders. .. setting:: LOG_DATEFORMAT @@ -912,10 +909,9 @@ LOG_DATEFORMAT Default: ``'%Y-%m-%d %H:%M:%S'`` String for formatting date/time, expansion of the ``%(asctime)s`` placeholder -in :setting:`LOG_FORMAT`. Refer to the `Python datetime documentation`_ for the whole list of available -directives. - -.. _Python datetime documentation: https://docs.python.org/3/library/datetime.html#strftime-and-strptime-behavior +in :setting:`LOG_FORMAT`. Refer to the +:ref:`Python datetime documentation <strftime-strptime-behavior>` for the +whole list of available directives. .. setting:: LOG_FORMATTER From 010edfe85caa72b4c366f2dada0f79f1f91e43ef Mon Sep 17 00:00:00 2001 From: Aditi Dutta <aditi011@e.ntu.edu.sg> Date: Wed, 25 Mar 2020 14:38:22 -0400 Subject: [PATCH 271/313] [Docs] mention curl2scrapy in Request.from_curl --- docs/topics/developer-tools.rst | 3 +++ docs/topics/dynamic-content.rst | 3 +++ scrapy/http/request/__init__.py | 3 +++ 3 files changed, 9 insertions(+) diff --git a/docs/topics/developer-tools.rst b/docs/topics/developer-tools.rst index f1b0964c6..4e87a00f2 100644 --- a/docs/topics/developer-tools.rst +++ b/docs/topics/developer-tools.rst @@ -292,6 +292,9 @@ Alternatively, if you want to know the arguments needed to recreate that request you can use the :func:`scrapy.utils.curl.curl_to_request_kwargs` function to get a dictionary with the equivalent arguments. +Note that to translate a cURL command into a Scrapy request, +you may use `curl2scrapy <https://michael-shub.github.io/curl2scrapy/>`_. + As you can see, with a few inspections in the `Network`-tool we were able to easily replicate the dynamic requests of the scrolling functionality of the page. Crawling dynamic pages can be quite diff --git a/docs/topics/dynamic-content.rst b/docs/topics/dynamic-content.rst index b98133676..aa326868b 100644 --- a/docs/topics/dynamic-content.rst +++ b/docs/topics/dynamic-content.rst @@ -104,6 +104,9 @@ If you get the expected response `sometimes`, but not always, the issue is probably not your request, but the target server. The target server might be buggy, overloaded, or :ref:`banning <bans>` some of your requests. +Note that to translate a cURL command into a Scrapy request, +you may use `curl2scrapy <https://michael-shub.github.io/curl2scrapy/>`_. + .. _topics-handling-response-formats: Handling different response formats diff --git a/scrapy/http/request/__init__.py b/scrapy/http/request/__init__.py index 6c536cb71..0a6637af8 100644 --- a/scrapy/http/request/__init__.py +++ b/scrapy/http/request/__init__.py @@ -129,6 +129,9 @@ class Request(object_ref): :class:`~scrapy.downloadermiddlewares.httpcompression.HttpCompressionMiddleware`, may modify the :class:`~scrapy.http.Request` object. + To translate a cURL command into a Scrapy request, + you may use `curl2scrapy <https://michael-shub.github.io/curl2scrapy/>`_. + """ request_kwargs = curl_to_request_kwargs(curl_command, ignore_unknown_options) request_kwargs.update(kwargs) From 16f2cb4a83033b63f3adf7d7cd8b7982b8b094ef Mon Sep 17 00:00:00 2001 From: Vostretsov Nikita <whalebot.helmsman@gmail.com> Date: Thu, 26 Mar 2020 12:57:39 +0000 Subject: [PATCH 272/313] project URLs in machine-readable format for showing in pypi --- setup.py | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/setup.py b/setup.py index 85d797f88..1b3c6771a 100644 --- a/setup.py +++ b/setup.py @@ -30,6 +30,11 @@ setup( name='Scrapy', version=version, url='https://scrapy.org', + project_urls = { + 'Documentation': 'https://docs.scrapy.org/', + 'Source': 'https://github.com/scrapy/scrapy', + 'Tracker': 'https://github.com/scrapy/scrapy/issues', + }, description='A high-level Web Crawling and Web Scraping framework', long_description=open('README.rst').read(), author='Scrapy developers', From b1904729d52d75bcf732b7ddccd7364e6efaa577 Mon Sep 17 00:00:00 2001 From: Aditya <k.aditya00@gmail.com> Date: Fri, 27 Mar 2020 04:37:26 +0530 Subject: [PATCH 273/313] [docs] change mod to doc redirect link --- docs/topics/request-response.rst | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 69a51a17c..573efc05f 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -174,7 +174,7 @@ Request objects See :ref:`topics-request-meta` for a list of special meta keys recognized by Scrapy. - This dict is :mod:`shallow copied <copy>` when the request is + This dict is :doc:`shallow copied <library/copy>` when the request is cloned using the ``copy()`` or ``replace()`` methods, and can also be accessed, in your spider, from the ``response.meta`` attribute. @@ -185,7 +185,7 @@ Request objects for new Requests, which means by default callbacks only get a :class:`Response` object as argument. - This dict is :mod:`shallow copied <copy>` when the request is + This dict is :doc:`shallow copied <library/copy>` when the request is cloned using the ``copy()`` or ``replace()`` methods, and can also be accessed, in your spider, from the ``response.cb_kwargs`` attribute. From e2d5d357a7ae50eaee957d1a2f8fc8ad1d9f3f24 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Wed, 1 Apr 2020 13:45:00 -0300 Subject: [PATCH 274/313] Fix pycodestyle E502 --- pytest.ini | 14 +++++++------- scrapy/commands/genspider.py | 7 +++---- scrapy/commands/shell.py | 2 +- scrapy/core/downloader/middleware.py | 18 ++++++++++++------ scrapy/core/engine.py | 5 ++--- scrapy/mail.py | 4 ++-- scrapy/utils/deprecate.py | 15 +++++++++------ tests/test_pipeline_media.py | 5 +++-- 8 files changed, 39 insertions(+), 31 deletions(-) diff --git a/pytest.ini b/pytest.ini index 141a13a4f..da0f68e20 100644 --- a/pytest.ini +++ b/pytest.ini @@ -36,24 +36,24 @@ flake8-ignore = scrapy/commands/crawl.py E501 scrapy/commands/edit.py E501 scrapy/commands/fetch.py E401 E501 E128 E731 - scrapy/commands/genspider.py E128 E501 E502 + scrapy/commands/genspider.py E128 E501 scrapy/commands/parse.py E128 E501 E731 scrapy/commands/runspider.py E501 scrapy/commands/settings.py E128 - scrapy/commands/shell.py E128 E501 E502 + scrapy/commands/shell.py E128 E501 scrapy/commands/startproject.py E127 E501 E128 scrapy/commands/version.py E501 E128 # scrapy/contracts scrapy/contracts/__init__.py E501 W504 scrapy/contracts/default.py E128 # scrapy/core - scrapy/core/engine.py E501 E128 E127 E502 + scrapy/core/engine.py E501 E128 E127 scrapy/core/scheduler.py E501 scrapy/core/scraper.py E501 E128 W504 scrapy/core/spidermw.py E501 E731 E126 scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 - scrapy/core/downloader/middleware.py E501 E502 + scrapy/core/downloader/middleware.py E501 scrapy/core/downloader/tls.py E501 E241 scrapy/core/downloader/webclient.py E731 E501 E128 E126 scrapy/core/downloader/handlers/__init__.py E501 @@ -124,7 +124,7 @@ flake8-ignore = scrapy/utils/datatypes.py E501 scrapy/utils/decorators.py E501 scrapy/utils/defer.py E501 E128 - scrapy/utils/deprecate.py E128 E501 E127 E502 + scrapy/utils/deprecate.py E128 E501 E127 scrapy/utils/gz.py E501 W504 scrapy/utils/http.py F403 scrapy/utils/httpobj.py E501 @@ -156,7 +156,7 @@ flake8-ignore = scrapy/item.py E501 E128 scrapy/link.py E501 scrapy/logformatter.py E501 - scrapy/mail.py E402 E128 E501 E502 + scrapy/mail.py E402 E128 E501 scrapy/middleware.py E128 E501 scrapy/pqueues.py E501 scrapy/resolver.py E501 @@ -214,7 +214,7 @@ flake8-ignore = tests/test_pipeline_crawl.py E501 E128 E126 tests/test_pipeline_files.py E501 tests/test_pipeline_images.py F841 E501 - tests/test_pipeline_media.py E501 E741 E731 E128 E502 + tests/test_pipeline_media.py E501 E741 E731 E128 tests/test_proxy_connect.py E501 E741 tests/test_request_cb_kwargs.py E501 tests/test_responsetypes.py E501 diff --git a/scrapy/commands/genspider.py b/scrapy/commands/genspider.py index adb01fa70..2e837abed 100644 --- a/scrapy/commands/genspider.py +++ b/scrapy/commands/genspider.py @@ -90,8 +90,7 @@ class Command(ScrapyCommand): 'module': module, 'name': name, 'domain': domain, - 'classname': '%sSpider' % ''.join(s.capitalize() \ - for s in module.split('_')) + 'classname': '%sSpider' % ''.join(s.capitalize() for s in module.split('_')) } if self.settings.get('NEWSPIDER_MODULE'): spiders_module = import_module(self.settings['NEWSPIDER_MODULE']) @@ -102,8 +101,8 @@ class Command(ScrapyCommand): spider_file = "%s.py" % join(spiders_dir, module) shutil.copyfile(template_file, spider_file) render_templatefile(spider_file, **tvars) - print("Created spider %r using template %r " % (name, \ - template_name), end=('' if spiders_module else '\n')) + print("Created spider %r using template %r " + % (name, template_name), end=('' if spiders_module else '\n')) if spiders_module: print("in module:\n %s.%s" % (spiders_module.__name__, module)) diff --git a/scrapy/commands/shell.py b/scrapy/commands/shell.py index d44a32d5f..5946f21e8 100644 --- a/scrapy/commands/shell.py +++ b/scrapy/commands/shell.py @@ -37,7 +37,7 @@ class Command(ScrapyCommand): help="evaluate the code in the shell, print the result and exit") parser.add_option("--spider", dest="spider", help="use this spider") - parser.add_option("--no-redirect", dest="no_redirect", action="store_true", \ + parser.add_option("--no-redirect", dest="no_redirect", action="store_true", default=False, help="do not handle HTTP 3xx status codes and print response as-is") def update_vars(self, vars): diff --git a/scrapy/core/downloader/middleware.py b/scrapy/core/downloader/middleware.py index 9c0014206..83c7b1f19 100644 --- a/scrapy/core/downloader/middleware.py +++ b/scrapy/core/downloader/middleware.py @@ -35,8 +35,10 @@ class DownloaderMiddlewareManager(MiddlewareManager): for method in self.methods['process_request']: response = yield deferred_from_coro(method(request=request, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): - raise _InvalidOutput('Middleware %s.process_request must return None, Response or Request, got %s' % \ - (method.__self__.__class__.__name__, response.__class__.__name__)) + raise _InvalidOutput( + "Middleware %s.process_request must return None, Response or Request, got %s" + % (method.__self__.__class__.__name__, response.__class__.__name__) + ) if response: defer.returnValue(response) defer.returnValue((yield download_func(request=request, spider=spider))) @@ -50,8 +52,10 @@ class DownloaderMiddlewareManager(MiddlewareManager): for method in self.methods['process_response']: response = yield deferred_from_coro(method(request=request, response=response, spider=spider)) if not isinstance(response, (Response, Request)): - raise _InvalidOutput('Middleware %s.process_response must return Response or Request, got %s' % \ - (method.__self__.__class__.__name__, type(response))) + raise _InvalidOutput( + "Middleware %s.process_response must return Response or Request, got %s" + % (method.__self__.__class__.__name__, type(response)) + ) if isinstance(response, Request): defer.returnValue(response) defer.returnValue(response) @@ -62,8 +66,10 @@ class DownloaderMiddlewareManager(MiddlewareManager): for method in self.methods['process_exception']: response = yield deferred_from_coro(method(request=request, exception=exception, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): - raise _InvalidOutput('Middleware %s.process_exception must return None, Response or Request, got %s' % \ - (method.__self__.__class__.__name__, type(response))) + raise _InvalidOutput( + "Middleware %s.process_exception must return None, Response or Request, got %s" + % (method.__self__.__class__.__name__, type(response)) + ) if response: defer.returnValue(response) defer.returnValue(_failure) diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 74f03344e..66cf9ad9a 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -277,10 +277,9 @@ class ExecutionEngine: next loop and this function is guaranteed to be called (at least) once again for this spider. """ - res = self.signals.send_catch_log(signal=signals.spider_idle, \ + res = self.signals.send_catch_log(signal=signals.spider_idle, spider=spider, dont_log=DontCloseSpider) - if any(isinstance(x, Failure) and isinstance(x.value, DontCloseSpider) \ - for _, x in res): + if any(isinstance(x, Failure) and isinstance(x.value, DontCloseSpider) for _, x in res): return if self.spider_is_idle(spider): diff --git a/scrapy/mail.py b/scrapy/mail.py index 9d186f4f3..9d7896ef6 100644 --- a/scrapy/mail.py +++ b/scrapy/mail.py @@ -115,8 +115,8 @@ class MailSender: from twisted.mail.smtp import ESMTPSenderFactory msg = BytesIO(msg) d = defer.Deferred() - factory = ESMTPSenderFactory(self.smtpuser, self.smtppass, self.mailfrom, \ - to_addrs, msg, d, heloFallback=True, requireAuthentication=False, \ + factory = ESMTPSenderFactory(self.smtpuser, self.smtppass, self.mailfrom, + to_addrs, msg, d, heloFallback=True, requireAuthentication=False, requireTransportSecurity=self.smtptls) factory.noisy = False diff --git a/scrapy/utils/deprecate.py b/scrapy/utils/deprecate.py index 69334a918..36001d982 100644 --- a/scrapy/utils/deprecate.py +++ b/scrapy/utils/deprecate.py @@ -7,9 +7,12 @@ from scrapy.exceptions import ScrapyDeprecationWarning def attribute(obj, oldattr, newattr, version='0.12'): cname = obj.__class__.__name__ - warnings.warn("%s.%s attribute is deprecated and will be no longer supported " - "in Scrapy %s, use %s.%s attribute instead" % \ - (cname, oldattr, version, cname, newattr), ScrapyDeprecationWarning, stacklevel=3) + warnings.warn( + "%s.%s attribute is deprecated and will be no longer supported " + "in Scrapy %s, use %s.%s attribute instead" + % (cname, oldattr, version, cname, newattr), + ScrapyDeprecationWarning, + stacklevel=3) def create_deprecated_class(name, new_class, clsdict=None, @@ -17,10 +20,10 @@ def create_deprecated_class(name, new_class, clsdict=None, warn_once=True, old_class_path=None, new_class_path=None, - subclass_warn_message="{cls} inherits from "\ - "deprecated class {old}, please inherit "\ + subclass_warn_message="{cls} inherits from " + "deprecated class {old}, please inherit " "from {new}.", - instance_warn_message="{cls} is deprecated, "\ + instance_warn_message="{cls} is deprecated, " "instantiate {new} instead."): """ Return a "deprecated" class that causes its subclasses to issue a warning. diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index d369e147d..ee1441227 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -325,8 +325,9 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): item = dict(requests=req) new_item = yield self.pipe.process_item(item, self.spider) self.assertEqual(new_item['results'], [(True, 'ITSME')]) - self.assertEqual(self.pipe._mockcalled, \ - ['get_media_requests', 'media_to_download', 'item_completed']) + self.assertEqual( + self.pipe._mockcalled, + ['get_media_requests', 'media_to_download', 'item_completed']) class MediaPipelineAllowRedirectSettingsTestCase(unittest.TestCase): From 4270e0a0da66a2cb3a8e904c5ea74f84b7f9d041 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sat, 4 Apr 2020 21:51:02 -0300 Subject: [PATCH 275/313] Fix E731: do not assign a lambda expression --- pytest.ini | 24 +++++++++++----------- scrapy/commands/fetch.py | 4 ++-- scrapy/commands/parse.py | 4 ++-- scrapy/core/downloader/webclient.py | 9 ++++---- scrapy/linkextractors/__init__.py | 10 +++++++-- scrapy/linkextractors/lxmlhtml.py | 6 ++---- tests/test_downloadermiddleware_cookies.py | 7 +++---- tests/test_exporters.py | 11 +++++----- tests/test_pipeline_media.py | 18 ++++++++++------ tests/test_utils_python.py | 4 +++- tests/test_utils_signal.py | 4 +++- 11 files changed, 57 insertions(+), 44 deletions(-) diff --git a/pytest.ini b/pytest.ini index 141a13a4f..4b655b8d5 100644 --- a/pytest.ini +++ b/pytest.ini @@ -35,9 +35,9 @@ flake8-ignore = scrapy/commands/check.py E501 scrapy/commands/crawl.py E501 scrapy/commands/edit.py E501 - scrapy/commands/fetch.py E401 E501 E128 E731 + scrapy/commands/fetch.py E401 E501 E128 scrapy/commands/genspider.py E128 E501 E502 - scrapy/commands/parse.py E128 E501 E731 + scrapy/commands/parse.py E128 E501 scrapy/commands/runspider.py E501 scrapy/commands/settings.py E128 scrapy/commands/shell.py E128 E501 E502 @@ -50,12 +50,12 @@ flake8-ignore = scrapy/core/engine.py E501 E128 E127 E502 scrapy/core/scheduler.py E501 scrapy/core/scraper.py E501 E128 W504 - scrapy/core/spidermw.py E501 E731 E126 + scrapy/core/spidermw.py E501 E126 scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 E502 scrapy/core/downloader/tls.py E501 E241 - scrapy/core/downloader/webclient.py E731 E501 E128 E126 + scrapy/core/downloader/webclient.py E501 E128 E126 scrapy/core/downloader/handlers/__init__.py E501 scrapy/core/downloader/handlers/ftp.py E501 E128 E127 scrapy/core/downloader/handlers/http10.py E501 @@ -90,8 +90,8 @@ flake8-ignore = scrapy/http/response/__init__.py E501 E128 scrapy/http/response/text.py E501 E128 E124 # scrapy/linkextractors - scrapy/linkextractors/__init__.py E731 E501 E402 W504 - scrapy/linkextractors/lxmlhtml.py E501 E731 + scrapy/linkextractors/__init__.py E501 E402 W504 + scrapy/linkextractors/lxmlhtml.py E501 # scrapy/loader scrapy/loader/__init__.py E501 E128 scrapy/loader/processors.py E501 @@ -184,7 +184,7 @@ flake8-ignore = tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 - tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126 + tests/test_downloadermiddleware_cookies.py E741 E501 E128 E265 E126 tests/test_downloadermiddleware_decompression.py E127 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 @@ -197,7 +197,7 @@ flake8-ignore = tests/test_downloadermiddleware_stats.py E501 tests/test_dupefilters.py E501 E741 E128 E124 tests/test_engine.py E401 E501 E128 - tests/test_exporters.py E501 E731 E128 E124 + tests/test_exporters.py E501 E128 E124 tests/test_extension_telnet.py F841 tests/test_feedexport.py E501 F841 E241 tests/test_http_cookies.py E501 @@ -207,14 +207,14 @@ flake8-ignore = tests/test_item.py E128 F841 tests/test_link.py E501 tests/test_linkextractors.py E501 E128 E124 - tests/test_loader.py E501 E731 E741 E128 E117 E241 + tests/test_loader.py E501 E741 E128 E117 E241 tests/test_logformatter.py E128 E501 E122 tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 tests/test_pipeline_crawl.py E501 E128 E126 tests/test_pipeline_files.py E501 tests/test_pipeline_images.py F841 E501 - tests/test_pipeline_media.py E501 E741 E731 E128 E502 + tests/test_pipeline_media.py E501 E741 E128 E502 tests/test_proxy_connect.py E501 E741 tests/test_request_cb_kwargs.py E501 tests/test_responsetypes.py E501 @@ -237,11 +237,11 @@ flake8-ignore = tests/test_utils_http.py E501 E128 W504 tests/test_utils_iterators.py E501 E128 E129 E241 tests/test_utils_log.py E741 - tests/test_utils_python.py E501 E731 + tests/test_utils_python.py E501 tests/test_utils_reqser.py E501 E128 tests/test_utils_request.py E501 E128 tests/test_utils_response.py E501 - tests/test_utils_signal.py E741 F841 E731 + tests/test_utils_signal.py E741 F841 tests/test_utils_sitemap.py E128 E501 E124 tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123 tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126 diff --git a/scrapy/commands/fetch.py b/scrapy/commands/fetch.py index 0e149941d..506d1f1b7 100644 --- a/scrapy/commands/fetch.py +++ b/scrapy/commands/fetch.py @@ -49,8 +49,8 @@ class Command(ScrapyCommand): def run(self, args, opts): if len(args) != 1 or not is_url(args[0]): raise UsageError() - cb = lambda x: self._print_response(x, opts) - request = Request(args[0], callback=cb, dont_filter=True) + request = Request(args[0], callback=self._print_response, + cb_kwargs={"opts": opts}, dont_filter=True) # by default, let the framework handle redirects, # i.e. command handles all codes expect 3xx if not opts.no_redirect: diff --git a/scrapy/commands/parse.py b/scrapy/commands/parse.py index 3ef8ddcb3..d5abe5930 100644 --- a/scrapy/commands/parse.py +++ b/scrapy/commands/parse.py @@ -147,8 +147,8 @@ class Command(ScrapyCommand): logger.error('Unable to find spider for: %(url)s', {'url': url}) # Request requires callback argument as callable or None, not string - request = Request(url, None) - _start_requests = lambda s: [self.prepare_request(s, request, opts)] + def _start_requests(spider): + yield self.prepare_request(spider, Request(url, None), opts) self.spidercls.start_requests = _start_requests def start_parsing(self, url, opts): diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index a71dc5fb3..a90a77b2b 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -14,13 +14,12 @@ from scrapy.responsetypes import responsetypes def _parsed_url_args(parsed): # Assume parsed is urlparse-d from Request.url, # which was passed via safe_url_string and is ascii-only. - b = lambda s: to_bytes(s, encoding='ascii') path = urlunparse(('', '', parsed.path or '/', parsed.params, parsed.query, '')) - path = b(path) - host = b(parsed.hostname) + path = to_bytes(path, encoding="ascii") + host = to_bytes(parsed.hostname, encoding="ascii") port = parsed.port - scheme = b(parsed.scheme) - netloc = b(parsed.netloc) + scheme = to_bytes(parsed.scheme, encoding="ascii") + netloc = to_bytes(parsed.netloc, encoding="ascii") if port is None: port = 443 if scheme == b'https' else 80 return scheme, netloc, host, port, path diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index 6afe867b5..d0b5066b6 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -45,8 +45,14 @@ IGNORED_EXTENSIONS = [ _re_type = type(re.compile("", 0)) -_matches = lambda url, regexs: any(r.search(url) for r in regexs) -_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'} + + +def _matches(url, regexs): + return any(r.search(url) for r in regexs) + + +def _is_valid_url(url): + return url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'} class FilteringLinkExtractor: diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index fbac1dc59..ceb37c5f1 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -98,11 +98,9 @@ class LxmlLinkExtractor(FilteringLinkExtractor): unique=True, process_value=None, deny_extensions=None, restrict_css=(), strip=True, restrict_text=None): tags, attrs = set(arg_to_iter(tags)), set(arg_to_iter(attrs)) - tag_func = lambda x: x in tags - attr_func = lambda x: x in attrs lx = LxmlParserLinkExtractor( - tag=tag_func, - attr=attr_func, + tag=lambda x: x in tags, + attr=lambda x: x in attrs, unique=unique, process=process_value, strip=strip, diff --git a/tests/test_downloadermiddleware_cookies.py b/tests/test_downloadermiddleware_cookies.py index 051f66680..a8182e2ef 100644 --- a/tests/test_downloadermiddleware_cookies.py +++ b/tests/test_downloadermiddleware_cookies.py @@ -13,10 +13,9 @@ from scrapy.downloadermiddlewares.cookies import CookiesMiddleware class CookiesMiddlewareTest(TestCase): def assertCookieValEqual(self, first, second, msg=None): - cookievaleq = lambda cv: re.split(r';\s*', cv.decode('latin1')) - return self.assertEqual( - sorted(cookievaleq(first)), - sorted(cookievaleq(second)), msg) + def split_cookies(cookies): + return sorted(re.split(r";\s*", cookies.decode("latin1"))) + return self.assertEqual(split_cookies(first), split_cookies(second), msg=msg) def setUp(self): self.spider = Spider('foo') diff --git a/tests/test_exporters.py b/tests/test_exporters.py index 6e2507508..160912847 100644 --- a/tests/test_exporters.py +++ b/tests/test_exporters.py @@ -215,11 +215,12 @@ class CsvItemExporterTest(BaseItemExporterTest): return CsvItemExporter(self.output, **kwargs) def assertCsvEqual(self, first, second, msg=None): - first = to_unicode(first) - second = to_unicode(second) - csvsplit = lambda csv: [sorted(re.split(r'(,|\s+)', line)) - for line in csv.splitlines(True)] - return self.assertEqual(csvsplit(first), csvsplit(second), msg) + def split_csv(csv): + return [ + sorted(re.split(r"(,|\s+)", line)) + for line in to_unicode(csv).splitlines(True) + ] + return self.assertEqual(split_csv(first), split_csv(second), msg=msg) def _check_output(self): self.assertCsvEqual(to_unicode(self.output.getvalue()), u'age,name\r\n22,John\xa3\r\n') diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index d369e147d..f84f47816 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -199,12 +199,19 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): pipeline_class = MockedMediaPipeline + def _callback(self, result): + self.pipe._mockcalled.append('request_callback') + return result + + def _errback(self, result): + self.pipe._mockcalled.append('request_errback') + return result + @inlineCallbacks def test_result_succeed(self): - cb = lambda _: self.pipe._mockcalled.append('request_callback') or _ - eb = lambda _: self.pipe._mockcalled.append('request_errback') or _ rsp = Response('http://url1') - req = Request('http://url1', meta=dict(response=rsp), callback=cb, errback=eb) + req = Request('http://url1', meta=dict(response=rsp), + callback=self._callback, errback=self._errback) item = dict(requests=req) new_item = yield self.pipe.process_item(item, self.spider) self.assertEqual(new_item['results'], [(True, rsp)]) @@ -215,10 +222,9 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase): @inlineCallbacks def test_result_failure(self): self.pipe.LOG_FAILED_RESULTS = False - cb = lambda _: self.pipe._mockcalled.append('request_callback') or _ - eb = lambda _: self.pipe._mockcalled.append('request_errback') or _ fail = Failure(Exception()) - req = Request('http://url1', meta=dict(response=fail), callback=cb, errback=eb) + req = Request('http://url1', meta=dict(response=fail), + callback=self._callback, errback=self._errback) item = dict(requests=req) new_item = yield self.pipe.process_item(item, self.spider) self.assertEqual(new_item['results'], [(False, fail)]) diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py index 8cb8df15b..65e6ba876 100644 --- a/tests/test_utils_python.py +++ b/tests/test_utils_python.py @@ -145,7 +145,9 @@ class UtilsPythonTestCase(unittest.TestCase): get_z = operator.itemgetter('z') get_meta = operator.attrgetter('meta') - compare_z = lambda obj: get_z(get_meta(obj)) + + def compare_z(obj): + return get_z(get_meta(obj)) self.assertTrue(equal_attributes(a, b, [compare_z, 'x'])) # fail z equality diff --git a/tests/test_utils_signal.py b/tests/test_utils_signal.py index 9f6da09ed..bb211dc60 100644 --- a/tests/test_utils_signal.py +++ b/tests/test_utils_signal.py @@ -90,8 +90,10 @@ class SendCatchLogDeferredAsyncioTest(SendCatchLogDeferredTest): class SendCatchLogTest2(unittest.TestCase): def test_error_logged_if_deferred_not_supported(self): + def test_handler(): + return defer.Deferred() + test_signal = object() - test_handler = lambda: defer.Deferred() dispatcher.connect(test_handler, test_signal) with LogCapture() as l: send_catch_log(test_signal) From c887fe37adfe529ed2afabd2d08c3aac00a819c0 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sat, 4 Apr 2020 22:15:36 -0300 Subject: [PATCH 276/313] Simplify parse command --- scrapy/commands/parse.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scrapy/commands/parse.py b/scrapy/commands/parse.py index d5abe5930..1cefed106 100644 --- a/scrapy/commands/parse.py +++ b/scrapy/commands/parse.py @@ -146,9 +146,8 @@ class Command(ScrapyCommand): if not self.spidercls: logger.error('Unable to find spider for: %(url)s', {'url': url}) - # Request requires callback argument as callable or None, not string def _start_requests(spider): - yield self.prepare_request(spider, Request(url, None), opts) + yield self.prepare_request(spider, Request(url), opts) self.spidercls.start_requests = _start_requests def start_parsing(self, url, opts): From 862f0301e2cb166ca87ce985d51683b46dcf56ad Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Sun, 5 Apr 2020 00:53:10 -0300 Subject: [PATCH 277/313] Remove empty _RequestBodyProducer for POST requests --- scrapy/core/downloader/handlers/http11.py | 14 -------------- 1 file changed, 14 deletions(-) diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index c970909d7..09f828419 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -341,20 +341,6 @@ class ScrapyAgent: headers.removeHeader(b'Proxy-Authorization') if request.body: bodyproducer = _RequestBodyProducer(request.body) - elif method == b'POST': - # Setting Content-Length: 0 even for POST requests is not a - # MUST per HTTP RFCs, but it's common behavior, and some - # servers require this, otherwise returning HTTP 411 Length required - # - # RFC 7230#section-3.3.2: - # "a Content-Length header field is normally sent in a POST - # request even when the value is 0 (indicating an empty payload body)." - # - # Twisted < 17 will not add "Content-Length: 0" by itself; - # Twisted >= 17 fixes this; - # Using a producer with an empty-string sends `0` as Content-Length - # for all versions of Twisted. - bodyproducer = _RequestBodyProducer(b'') else: bodyproducer = None start_time = time() From f97fec5ebd34776fdd97220cb2b1eff1f639a409 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Apr 2020 16:02:53 -0300 Subject: [PATCH 278/313] Pin Sphinx version, including extensions --- docs/requirements.txt | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/requirements.txt b/docs/requirements.txt index 773b92cea..a99d1b78f 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1,4 +1,4 @@ -Sphinx>=2.1 -sphinx-hoverxref -sphinx-notfound-page -sphinx_rtd_theme +Sphinx==3.0.1 +sphinx-hoverxref==0.2b1 +sphinx-notfound-page==0.4 +sphinx_rtd_theme==0.4.3 From 24a1d9acae776bc195e6078394ee159b42275833 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Apr 2020 16:48:42 -0300 Subject: [PATCH 279/313] Get version in docs config --- docs/conf.py | 11 +++++++---- 1 file changed, 7 insertions(+), 4 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 6e2399f66..c59688fbe 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -14,6 +14,7 @@ import sys from datetime import datetime from os import path +from pathlib import Path # If your extensions are in another directory, add it here. If the directory # is relative to the documentation root, use os.path.abspath to make it @@ -59,10 +60,10 @@ copyright = '2008–{}, Scrapy developers'.format(datetime.now().year) # # The short X.Y version. try: - import scrapy - version = '.'.join(map(str, scrapy.version_info[:2])) - release = scrapy.__version__ -except ImportError: + version_path = Path(__file__).parent.absolute().parent.joinpath("scrapy/VERSION") + version = version_path.read_text().strip() + release = version.rsplit(".", 1)[0] +except Exception: version = '' release = '' @@ -295,3 +296,5 @@ intersphinx_mapping = { # ------------------------------------ hoverxref_auto_ref = True +hoverxref_project = "scrapy" +hoverxref_version = release From 4383f452999464b623393288361ecf7f383666e2 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Apr 2020 16:49:14 -0300 Subject: [PATCH 280/313] Replace os.path with pathlib in docs config --- docs/conf.py | 5 ++--- 1 file changed, 2 insertions(+), 3 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index c59688fbe..a0bbbc90a 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -13,14 +13,13 @@ import sys from datetime import datetime -from os import path from pathlib import Path # If your extensions are in another directory, add it here. If the directory # is relative to the documentation root, use os.path.abspath to make it # absolute, like shown here. -sys.path.append(path.join(path.dirname(__file__), "_ext")) -sys.path.insert(0, path.dirname(path.dirname(__file__))) +sys.path.append(str(Path(__file__).absolute().parent / "_ext")) +sys.path.insert(0, str(Path(__file__).absolute().parent.parent)) # General configuration From 34e81d0d74fb0c5b9e880afdf214c2fd2ec193c6 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Apr 2020 17:29:02 -0300 Subject: [PATCH 281/313] Docs: remove duplicated setting definitions --- docs/topics/downloader-middleware.rst | 1 + docs/topics/settings.rst | 22 ---------------------- 2 files changed, 1 insertion(+), 22 deletions(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 73648994d..cea5e4564 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -829,6 +829,7 @@ REDIRECT_MAX_TIMES Default: ``20`` The maximum number of redirections that will be followed for a single request. +After this maximum the request's response is returned as is. MetaRefreshMiddleware --------------------- diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index dc6843d75..90df9a02e 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -1116,17 +1116,6 @@ multi-purpose thread pool used by various Scrapy components. Threaded DNS Resolver, BlockingFeedStorage, S3FilesStore just to name a few. Increase this value if you're experiencing problems with insufficient blocking IO. -.. setting:: REDIRECT_MAX_TIMES - -REDIRECT_MAX_TIMES ------------------- - -Default: ``20`` - -Defines the maximum times a request can be redirected. After this maximum the -request's response is returned as is. We used Firefox default value for the -same task. - .. setting:: REDIRECT_PRIORITY_ADJUST REDIRECT_PRIORITY_ADJUST @@ -1422,17 +1411,6 @@ Default: ``True`` A boolean which specifies if the :ref:`telnet console <topics-telnetconsole>` will be enabled (provided its extension is also enabled). -.. setting:: TELNETCONSOLE_PORT - -TELNETCONSOLE_PORT ------------------- - -Default: ``[6023, 6073]`` - -The port range to use for the telnet console. If set to ``None`` or ``0``, a -dynamically assigned port is used. For more info see -:ref:`topics-telnetconsole`. - .. setting:: TEMPLATES_DIR TEMPLATES_DIR From 2205f04631d97103f98a28f865e8ac6511c15c82 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Fri, 10 Apr 2020 18:08:04 -0300 Subject: [PATCH 282/313] Docs: Add hoverxref_role_types setting --- docs/conf.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/docs/conf.py b/docs/conf.py index a0bbbc90a..40de81342 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -297,3 +297,10 @@ intersphinx_mapping = { hoverxref_auto_ref = True hoverxref_project = "scrapy" hoverxref_version = release +hoverxref_role_types = { + "class": "tooltip", + "confval": "tooltip", + "hoverxref": "tooltip", + "mod": "tooltip", + "ref": "tooltip", +} From 1bd8f392c92d5e856332ab99f547d2a4359bd5d1 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 13 Apr 2020 06:12:30 -0300 Subject: [PATCH 283/313] Initial removal of twisted.internet.defer.returnValue --- scrapy/core/downloader/middleware.py | 18 +++++++++--------- tests/test_feedexport.py | 6 +++--- tests/test_spidermiddleware_output_chain.py | 2 +- 3 files changed, 13 insertions(+), 13 deletions(-) diff --git a/scrapy/core/downloader/middleware.py b/scrapy/core/downloader/middleware.py index 83c7b1f19..5a03dcdf7 100644 --- a/scrapy/core/downloader/middleware.py +++ b/scrapy/core/downloader/middleware.py @@ -40,14 +40,14 @@ class DownloaderMiddlewareManager(MiddlewareManager): % (method.__self__.__class__.__name__, response.__class__.__name__) ) if response: - defer.returnValue(response) - defer.returnValue((yield download_func(request=request, spider=spider))) + return response + return (yield download_func(request=request, spider=spider)) @defer.inlineCallbacks def process_response(response): assert response is not None, 'Received None in process_response' if isinstance(response, Request): - defer.returnValue(response) + return response for method in self.methods['process_response']: response = yield deferred_from_coro(method(request=request, response=response, spider=spider)) @@ -57,12 +57,12 @@ class DownloaderMiddlewareManager(MiddlewareManager): % (method.__self__.__class__.__name__, type(response)) ) if isinstance(response, Request): - defer.returnValue(response) - defer.returnValue(response) + return response + return response @defer.inlineCallbacks - def process_exception(_failure): - exception = _failure.value + def process_exception(failure): + exception = failure.value for method in self.methods['process_exception']: response = yield deferred_from_coro(method(request=request, exception=exception, spider=spider)) if response is not None and not isinstance(response, (Response, Request)): @@ -71,8 +71,8 @@ class DownloaderMiddlewareManager(MiddlewareManager): % (method.__self__.__class__.__name__, type(response)) ) if response: - defer.returnValue(response) - defer.returnValue(_failure) + return response + return failure deferred = mustbe_deferred(process_request, request) deferred.addErrback(process_exception) diff --git a/tests/test_feedexport.py b/tests/test_feedexport.py index c5589e52f..e02b0b840 100644 --- a/tests/test_feedexport.py +++ b/tests/test_feedexport.py @@ -433,7 +433,7 @@ class FeedExportTest(unittest.TestCase): for file_path in FEEDS.keys(): os.remove(str(file_path)) - defer.returnValue(content) + return content @defer.inlineCallbacks def exported_data(self, items, settings): @@ -448,7 +448,7 @@ class FeedExportTest(unittest.TestCase): yield item data = yield self.run_and_export(TestSpider, settings) - defer.returnValue(data) + return data @defer.inlineCallbacks def exported_no_data(self, settings): @@ -462,7 +462,7 @@ class FeedExportTest(unittest.TestCase): pass data = yield self.run_and_export(TestSpider, settings) - defer.returnValue(data) + return data @defer.inlineCallbacks def assertExportedCsv(self, items, header, rows, settings=None, ordered=True): diff --git a/tests/test_spidermiddleware_output_chain.py b/tests/test_spidermiddleware_output_chain.py index b26353d6c..ad4d6fb98 100644 --- a/tests/test_spidermiddleware_output_chain.py +++ b/tests/test_spidermiddleware_output_chain.py @@ -292,7 +292,7 @@ class TestSpiderMiddleware(TestCase): crawler = get_crawler(spider) with LogCapture() as log: yield crawler.crawl(mockserver=self.mockserver) - raise defer.returnValue(log) + return log @defer.inlineCallbacks def test_recovery(self): From 4023d5db33b588b4df861581948e39b41c0d1678 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Mon, 13 Apr 2020 06:35:17 -0300 Subject: [PATCH 284/313] Replace _DefGen_Return exception handling Handle StopIteration instead --- scrapy/pipelines/media.py | 24 +++++++++++++----------- tests/test_pipeline_media.py | 11 +++++------ 2 files changed, 18 insertions(+), 17 deletions(-) diff --git a/scrapy/pipelines/media.py b/scrapy/pipelines/media.py index 562d9ee32..a31c37900 100644 --- a/scrapy/pipelines/media.py +++ b/scrapy/pipelines/media.py @@ -1,7 +1,7 @@ import functools import logging from collections import defaultdict -from twisted.internet.defer import Deferred, DeferredList, _DefGen_Return +from twisted.internet.defer import Deferred, DeferredList from twisted.python.failure import Failure from scrapy.settings import Settings @@ -141,24 +141,26 @@ class MediaPipeline: # This code fixes a memory leak by avoiding to keep references to # the Request and Response objects on the Media Pipeline cache. # - # Twisted inline callbacks pass return values using the function - # twisted.internet.defer.returnValue, which encapsulates the return - # value inside a _DefGen_Return base exception. - # - # What happens when the media_downloaded callback raises another + # What happens when the media_downloaded callback raises an # exception, for example a FileException('download-error') when - # the Response status code is not 200 OK, is that it stores the - # _DefGen_Return exception on the FileException context. + # the Response status code is not 200 OK, is that the original + # StopIteration exception (which in turn contains the failed + # Response and by extension, the original Request) gets encapsulated + # within the FileException context. + # + # Originally, Scrapy was using twisted.internet.defer.returnValue + # inside functions decorated with twisted.internet.defer.inlineCallbacks, + # encapsulating the returned Response in a _DefGen_Return exception + # instead of a StopIteration. # # To avoid keeping references to the Response and therefore Request # objects on the Media Pipeline cache, we should wipe the context of - # the exception encapsulated by the Twisted Failure when its a - # _DefGen_Return instance. + # the encapsulated exception when it is a StopIteration instance # # This problem does not occur in Python 2.7 since we don't have # Exception Chaining (https://www.python.org/dev/peps/pep-3134/). context = getattr(result.value, '__context__', None) - if isinstance(context, _DefGen_Return): + if isinstance(context, StopIteration): setattr(result.value, '__context__', None) info.downloading.remove(fp) diff --git a/tests/test_pipeline_media.py b/tests/test_pipeline_media.py index ee1441227..e6e21601b 100644 --- a/tests/test_pipeline_media.py +++ b/tests/test_pipeline_media.py @@ -2,7 +2,7 @@ from testfixtures import LogCapture from twisted.trial import unittest from twisted.python.failure import Failure from twisted.internet import reactor -from twisted.internet.defer import Deferred, inlineCallbacks, returnValue +from twisted.internet.defer import Deferred, inlineCallbacks from scrapy.http import Request, Response from scrapy.settings import Settings @@ -124,9 +124,8 @@ class BaseMediaPipelineTestCase(unittest.TestCase): # Simulate the Media Pipeline behavior to produce a Twisted Failure try: # Simulate a Twisted inline callback returning a Response - # The returnValue method raises an exception encapsulating the value - returnValue(response) - except BaseException as exc: + raise StopIteration(response) + except StopIteration as exc: def_gen_return_exc = exc try: # Simulate the media_downloaded callback raising a FileException @@ -140,7 +139,7 @@ class BaseMediaPipelineTestCase(unittest.TestCase): # The Failure should encapsulate a FileException ... self.assertEqual(failure.value, file_exc) - # ... and it should have the returnValue exception set as its context + # ... and it should have the StopIteration exception set as its context self.assertEqual(failure.value.__context__, def_gen_return_exc) # Let's calculate the request fingerprint and fake some runtime data... @@ -155,7 +154,7 @@ class BaseMediaPipelineTestCase(unittest.TestCase): self.assertEqual(info.downloaded[fp], failure) # ... encapsulating the original FileException ... self.assertEqual(info.downloaded[fp].value, file_exc) - # ... but it should not store the returnValue exception on its context + # ... but it should not store the StopIteration exception on its context context = getattr(info.downloaded[fp].value, '__context__', None) self.assertIsNone(context) From 0a4ef97fa3d9d25b3f6b4afcf2b4986c505605c9 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Tue, 14 Apr 2020 14:57:20 -0300 Subject: [PATCH 285/313] Loose restrictions for docs requirements --- docs/requirements.txt | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/requirements.txt b/docs/requirements.txt index a99d1b78f..3d34b47da 100644 --- a/docs/requirements.txt +++ b/docs/requirements.txt @@ -1,4 +1,4 @@ -Sphinx==3.0.1 -sphinx-hoverxref==0.2b1 -sphinx-notfound-page==0.4 -sphinx_rtd_theme==0.4.3 +Sphinx>=3.0 +sphinx-hoverxref>=0.2b1 +sphinx-notfound-page>=0.4 +sphinx_rtd_theme>=0.4 From ee4ee486b1b4f66deffccbbe15f056edaf135982 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <eugenio.lacuesta@gmail.com> Date: Tue, 14 Apr 2020 15:06:54 -0300 Subject: [PATCH 286/313] Revert unnecessary changes to docs/conf.py --- docs/conf.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 40de81342..4414ef637 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -13,13 +13,13 @@ import sys from datetime import datetime -from pathlib import Path +from os import path # If your extensions are in another directory, add it here. If the directory # is relative to the documentation root, use os.path.abspath to make it # absolute, like shown here. -sys.path.append(str(Path(__file__).absolute().parent / "_ext")) -sys.path.insert(0, str(Path(__file__).absolute().parent.parent)) +sys.path.append(path.join(path.dirname(__file__), "_ext")) +sys.path.insert(0, path.dirname(path.dirname(__file__))) # General configuration @@ -59,10 +59,10 @@ copyright = '2008–{}, Scrapy developers'.format(datetime.now().year) # # The short X.Y version. try: - version_path = Path(__file__).parent.absolute().parent.joinpath("scrapy/VERSION") - version = version_path.read_text().strip() - release = version.rsplit(".", 1)[0] -except Exception: + import scrapy + version = '.'.join(map(str, scrapy.version_info[:2])) + release = scrapy.__version__ +except ImportError: version = '' release = '' From 94d7ad76cb96f1623d5944c28db24744955103cd Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <elacuesta@users.noreply.github.com> Date: Wed, 15 Apr 2020 09:11:37 -0300 Subject: [PATCH 287/313] Fix pycodestyle E2XX (whitespace) (#4468) --- pytest.ini | 30 +- scrapy/core/downloader/tls.py | 4 +- scrapy/dupefilters.py | 2 +- scrapy/pipelines/files.py | 2 +- scrapy/pipelines/images.py | 2 +- scrapy/pipelines/media.py | 2 +- tests/test_crawl.py | 6 +- tests/test_downloadermiddleware_cookies.py | 4 +- tests/test_http_response.py | 4 +- tests/test_spidermiddleware_referer.py | 333 +++++++++++---------- tests/test_utils_iterators.py | 50 ++-- tests/test_utils_url.py | 40 +-- tests/test_webclient.py | 30 +- 13 files changed, 268 insertions(+), 241 deletions(-) diff --git a/pytest.ini b/pytest.ini index da0f68e20..de0bccbf1 100644 --- a/pytest.ini +++ b/pytest.ini @@ -54,7 +54,7 @@ flake8-ignore = scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 scrapy/core/downloader/middleware.py E501 - scrapy/core/downloader/tls.py E501 E241 + scrapy/core/downloader/tls.py E501 scrapy/core/downloader/webclient.py E731 E501 E128 E126 scrapy/core/downloader/handlers/__init__.py E501 scrapy/core/downloader/handlers/ftp.py E501 E128 E127 @@ -97,9 +97,9 @@ flake8-ignore = scrapy/loader/processors.py E501 # scrapy/pipelines scrapy/pipelines/__init__.py E501 - scrapy/pipelines/files.py E116 E501 E266 - scrapy/pipelines/images.py E265 E501 - scrapy/pipelines/media.py E125 E501 E266 + scrapy/pipelines/files.py E116 E501 + scrapy/pipelines/images.py E501 + scrapy/pipelines/media.py E125 E501 # scrapy/selector scrapy/selector/__init__.py F403 scrapy/selector/unified.py E501 E111 @@ -149,7 +149,7 @@ flake8-ignore = scrapy/__init__.py E402 E501 scrapy/cmdline.py E501 scrapy/crawler.py E501 - scrapy/dupefilters.py E501 E202 + scrapy/dupefilters.py E501 scrapy/exceptions.py E501 scrapy/exporters.py E501 scrapy/interfaces.py E501 @@ -178,13 +178,13 @@ flake8-ignore = tests/test_command_shell.py E501 E128 tests/test_commands.py E128 E501 tests/test_contracts.py E501 E128 - tests/test_crawl.py E501 E741 E265 + tests/test_crawl.py E501 E741 tests/test_crawler.py F841 E501 tests/test_dependencies.py F841 E501 - tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123 + tests/test_downloader_handlers.py E124 E127 E128 E501 E126 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 - tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126 + tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E126 tests/test_downloadermiddleware_decompression.py E127 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 @@ -199,15 +199,15 @@ flake8-ignore = tests/test_engine.py E401 E501 E128 tests/test_exporters.py E501 E731 E128 E124 tests/test_extension_telnet.py F841 - tests/test_feedexport.py E501 F841 E241 + tests/test_feedexport.py E501 F841 tests/test_http_cookies.py E501 tests/test_http_headers.py E501 tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 - tests/test_http_response.py E501 E128 E265 + tests/test_http_response.py E501 E128 tests/test_item.py E128 F841 tests/test_link.py E501 tests/test_linkextractors.py E501 E128 E124 - tests/test_loader.py E501 E731 E741 E128 E117 E241 + tests/test_loader.py E501 E731 E741 E128 E117 tests/test_logformatter.py E128 E501 E122 tests/test_mail.py E128 E501 tests/test_middleware.py E501 E128 @@ -226,7 +226,7 @@ flake8-ignore = tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 tests/test_spidermiddleware_offsite.py E501 E128 E111 tests/test_spidermiddleware_output_chain.py E501 - tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121 + tests/test_spidermiddleware_referer.py E501 F841 E125 E124 E501 E121 tests/test_squeues.py E501 E741 tests/test_utils_asyncio.py E501 tests/test_utils_conf.py E501 E128 @@ -235,7 +235,7 @@ flake8-ignore = tests/test_utils_defer.py E501 F841 tests/test_utils_deprecate.py F841 E501 tests/test_utils_http.py E501 E128 W504 - tests/test_utils_iterators.py E501 E128 E129 E241 + tests/test_utils_iterators.py E501 E128 E129 tests/test_utils_log.py E741 tests/test_utils_python.py E501 E731 tests/test_utils_reqser.py E501 E128 @@ -243,8 +243,8 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 E731 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123 - tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126 + tests/test_utils_url.py E501 E127 E125 E501 E126 E123 + tests/test_webclient.py E501 E128 E122 E402 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 tests/test_spiderloader/__init__.py E128 E501 diff --git a/scrapy/core/downloader/tls.py b/scrapy/core/downloader/tls.py index a1c881d5e..e43a3c83e 100644 --- a/scrapy/core/downloader/tls.py +++ b/scrapy/core/downloader/tls.py @@ -20,8 +20,8 @@ METHOD_TLSv12 = 'TLSv1.2' openssl_methods = { - METHOD_TLS: SSL.SSLv23_METHOD, # protocol negotiation (recommended) - METHOD_SSLv3: SSL.SSLv3_METHOD, # SSL 3 (NOT recommended) + METHOD_TLS: SSL.SSLv23_METHOD, # protocol negotiation (recommended) + METHOD_SSLv3: SSL.SSLv3_METHOD, # SSL 3 (NOT recommended) METHOD_TLSv10: SSL.TLSv1_METHOD, # TLS 1.0 only METHOD_TLSv11: getattr(SSL, 'TLSv1_1_METHOD', 5), # TLS 1.1 only METHOD_TLSv12: getattr(SSL, 'TLSv1_2_METHOD', 6), # TLS 1.2 only diff --git a/scrapy/dupefilters.py b/scrapy/dupefilters.py index d74c8ed36..ac5478e7c 100644 --- a/scrapy/dupefilters.py +++ b/scrapy/dupefilters.py @@ -61,7 +61,7 @@ class RFPDupeFilter(BaseDupeFilter): def log(self, request, spider): if self.debug: msg = "Filtered duplicate request: %(request)s (referer: %(referer)s)" - args = {'request': request, 'referer': referer_str(request) } + args = {'request': request, 'referer': referer_str(request)} self.logger.debug(msg, args, extra={'spider': spider}) elif self.logdupes: msg = ("Filtered duplicate request: %(request)s" diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index 101bf5fbc..aab645d3d 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -500,7 +500,7 @@ class FilesPipeline(MediaPipeline): spider.crawler.stats.inc_value('file_count', spider=spider) spider.crawler.stats.inc_value('file_status_count/%s' % status, spider=spider) - ### Overridable Interface + # Overridable Interface def get_media_requests(self, item, info): return [Request(x) for x in item.get(self.files_urls_field, [])] diff --git a/scrapy/pipelines/images.py b/scrapy/pipelines/images.py index 2e646379c..aeb520442 100644 --- a/scrapy/pipelines/images.py +++ b/scrapy/pipelines/images.py @@ -14,7 +14,7 @@ from scrapy.utils.python import to_bytes from scrapy.http import Request from scrapy.settings import Settings from scrapy.exceptions import DropItem -#TODO: from scrapy.pipelines.media import MediaPipeline +# TODO: from scrapy.pipelines.media import MediaPipeline from scrapy.pipelines.files import FileException, FilesPipeline diff --git a/scrapy/pipelines/media.py b/scrapy/pipelines/media.py index 562d9ee32..a6d99fa99 100644 --- a/scrapy/pipelines/media.py +++ b/scrapy/pipelines/media.py @@ -166,7 +166,7 @@ class MediaPipeline: for wad in info.waiting.pop(fp): defer_result(result).chainDeferred(wad) - ### Overridable Interface + # Overridable Interface def media_to_download(self, request, info): """Check request before starting download""" pass diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 3f8a7435c..c02e6a70b 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -147,9 +147,9 @@ class CrawlTestCase(TestCase): settings = {"CONCURRENT_REQUESTS": 1} crawler = CrawlerRunner(settings).create_crawler(BrokenStartRequestsSpider) yield crawler.crawl(mockserver=self.mockserver) - #self.assertTrue(False, crawler.spider.seedsseen) - #self.assertTrue(crawler.spider.seedsseen.index(None) < crawler.spider.seedsseen.index(99), - # crawler.spider.seedsseen) + self.assertTrue( + crawler.spider.seedsseen.index(None) < crawler.spider.seedsseen.index(99), + crawler.spider.seedsseen) @defer.inlineCallbacks def test_start_requests_dupes(self): diff --git a/tests/test_downloadermiddleware_cookies.py b/tests/test_downloadermiddleware_cookies.py index 051f66680..f8e4851fc 100644 --- a/tests/test_downloadermiddleware_cookies.py +++ b/tests/test_downloadermiddleware_cookies.py @@ -202,7 +202,7 @@ class CookiesMiddlewareTest(TestCase): assert self.mw.process_request(req4, self.spider) is None self.assertCookieValEqual(req4.headers.get('Cookie'), b'C2=value2; galleta=dulce') - #cookies from hosts with port + # cookies from hosts with port req5_1 = Request('http://scrapytest.org:1104/') assert self.mw.process_request(req5_1, self.spider) is None @@ -218,7 +218,7 @@ class CookiesMiddlewareTest(TestCase): assert self.mw.process_request(req5_3, self.spider) is None self.assertEqual(req5_3.headers.get('Cookie'), b'C1=value1') - #skip cookie retrieval for not http request + # skip cookie retrieval for not http request req6 = Request('file:///scrapy/sometempfile') assert self.mw.process_request(req6, self.spider) is None self.assertEqual(req6.headers.get('Cookie'), None) diff --git a/tests/test_http_response.py b/tests/test_http_response.py index eafc3560e..522ec4875 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -438,8 +438,8 @@ class TextResponseTest(BaseResponseTest): assert u'<span>value</span>' in r.text, repr(r.text) # FIXME: This test should pass once we stop using BeautifulSoup's UnicodeDammit in TextResponse - #r = self.response_class("http://www.example.com", body=b'PREFIX\xe3\xabSUFFIX') - #assert u'\ufffd' in r.text, repr(r.text) + # r = self.response_class("http://www.example.com", body=b'PREFIX\xe3\xabSUFFIX') + # assert u'\ufffd' in r.text, repr(r.text) def test_selector(self): body = b"<html><head><title>Some page" diff --git a/tests/test_spidermiddleware_referer.py b/tests/test_spidermiddleware_referer.py index 4c6ede70b..742adc64f 100644 --- a/tests/test_spidermiddleware_referer.py +++ b/tests/test_spidermiddleware_referer.py @@ -24,7 +24,7 @@ class TestRefererMiddleware(TestCase): resp_headers = {} settings = {} scenarii = [ - ('http://scrapytest.org', 'http://scrapytest.org/', b'http://scrapytest.org'), + ('http://scrapytest.org', 'http://scrapytest.org/', b'http://scrapytest.org'), ] def setUp(self): @@ -54,57 +54,57 @@ class MixinDefault: with some additional filtering of s3:// """ scenarii = [ - ('https://example.com/', 'https://scrapy.org/', b'https://example.com/'), - ('http://example.com/', 'http://scrapy.org/', b'http://example.com/'), - ('http://example.com/', 'https://scrapy.org/', b'http://example.com/'), - ('https://example.com/', 'http://scrapy.org/', None), + ('https://example.com/', 'https://scrapy.org/', b'https://example.com/'), + ('http://example.com/', 'http://scrapy.org/', b'http://example.com/'), + ('http://example.com/', 'https://scrapy.org/', b'http://example.com/'), + ('https://example.com/', 'http://scrapy.org/', None), # no credentials leak - ('http://user:password@example.com/', 'https://scrapy.org/', b'http://example.com/'), + ('http://user:password@example.com/', 'https://scrapy.org/', b'http://example.com/'), # no referrer leak for local schemes - ('file:///home/path/to/somefile.html', 'https://scrapy.org/', None), - ('file:///home/path/to/somefile.html', 'http://scrapy.org/', None), + ('file:///home/path/to/somefile.html', 'https://scrapy.org/', None), + ('file:///home/path/to/somefile.html', 'http://scrapy.org/', None), # no referrer leak for s3 origins - ('s3://mybucket/path/to/data.csv', 'https://scrapy.org/', None), - ('s3://mybucket/path/to/data.csv', 'http://scrapy.org/', None), + ('s3://mybucket/path/to/data.csv', 'https://scrapy.org/', None), + ('s3://mybucket/path/to/data.csv', 'http://scrapy.org/', None), ] class MixinNoReferrer: scenarii = [ - ('https://example.com/page.html', 'https://example.com/', None), - ('http://www.example.com/', 'https://scrapy.org/', None), - ('http://www.example.com/', 'http://scrapy.org/', None), - ('https://www.example.com/', 'http://scrapy.org/', None), - ('file:///home/path/to/somefile.html', 'http://scrapy.org/', None), + ('https://example.com/page.html', 'https://example.com/', None), + ('http://www.example.com/', 'https://scrapy.org/', None), + ('http://www.example.com/', 'http://scrapy.org/', None), + ('https://www.example.com/', 'http://scrapy.org/', None), + ('file:///home/path/to/somefile.html', 'http://scrapy.org/', None), ] class MixinNoReferrerWhenDowngrade: scenarii = [ # TLS to TLS: send non-empty referrer - ('https://example.com/page.html', 'https://not.example.com/', b'https://example.com/page.html'), - ('https://example.com/page.html', 'https://scrapy.org/', b'https://example.com/page.html'), - ('https://example.com:443/page.html', 'https://scrapy.org/', b'https://example.com/page.html'), - ('https://example.com:444/page.html', 'https://scrapy.org/', b'https://example.com:444/page.html'), - ('ftps://example.com/urls.zip', 'https://scrapy.org/', b'ftps://example.com/urls.zip'), + ('https://example.com/page.html', 'https://not.example.com/', b'https://example.com/page.html'), + ('https://example.com/page.html', 'https://scrapy.org/', b'https://example.com/page.html'), + ('https://example.com:443/page.html', 'https://scrapy.org/', b'https://example.com/page.html'), + ('https://example.com:444/page.html', 'https://scrapy.org/', b'https://example.com:444/page.html'), + ('ftps://example.com/urls.zip', 'https://scrapy.org/', b'ftps://example.com/urls.zip'), # TLS to non-TLS: do not send referrer - ('https://example.com/page.html', 'http://not.example.com/', None), - ('https://example.com/page.html', 'http://scrapy.org/', None), - ('ftps://example.com/urls.zip', 'http://scrapy.org/', None), + ('https://example.com/page.html', 'http://not.example.com/', None), + ('https://example.com/page.html', 'http://scrapy.org/', None), + ('ftps://example.com/urls.zip', 'http://scrapy.org/', None), # non-TLS to TLS or non-TLS: send referrer - ('http://example.com/page.html', 'https://not.example.com/', b'http://example.com/page.html'), - ('http://example.com/page.html', 'https://scrapy.org/', b'http://example.com/page.html'), - ('http://example.com:8080/page.html', 'https://scrapy.org/', b'http://example.com:8080/page.html'), - ('http://example.com:80/page.html', 'http://not.example.com/', b'http://example.com/page.html'), - ('http://example.com/page.html', 'http://scrapy.org/', b'http://example.com/page.html'), - ('http://example.com:443/page.html', 'http://scrapy.org/', b'http://example.com:443/page.html'), - ('ftp://example.com/urls.zip', 'http://scrapy.org/', b'ftp://example.com/urls.zip'), - ('ftp://example.com/urls.zip', 'https://scrapy.org/', b'ftp://example.com/urls.zip'), + ('http://example.com/page.html', 'https://not.example.com/', b'http://example.com/page.html'), + ('http://example.com/page.html', 'https://scrapy.org/', b'http://example.com/page.html'), + ('http://example.com:8080/page.html', 'https://scrapy.org/', b'http://example.com:8080/page.html'), + ('http://example.com:80/page.html', 'http://not.example.com/', b'http://example.com/page.html'), + ('http://example.com/page.html', 'http://scrapy.org/', b'http://example.com/page.html'), + ('http://example.com:443/page.html', 'http://scrapy.org/', b'http://example.com:443/page.html'), + ('ftp://example.com/urls.zip', 'http://scrapy.org/', b'ftp://example.com/urls.zip'), + ('ftp://example.com/urls.zip', 'https://scrapy.org/', b'ftp://example.com/urls.zip'), # test for user/password stripping ('http://user:password@example.com/page.html', 'https://not.example.com/', b'http://example.com/page.html'), @@ -114,43 +114,43 @@ class MixinNoReferrerWhenDowngrade: class MixinSameOrigin: scenarii = [ # Same origin (protocol, host, port): send referrer - ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('http://example.com/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), - ('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), - ('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'), - ('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'), + ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('http://example.com/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), + ('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), + ('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'), + ('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'), # Different host: do NOT send referrer - ('https://example.com/page.html', 'https://not.example.com/otherpage.html', None), - ('http://example.com/page.html', 'http://not.example.com/otherpage.html', None), - ('http://example.com/page.html', 'http://www.example.com/otherpage.html', None), + ('https://example.com/page.html', 'https://not.example.com/otherpage.html', None), + ('http://example.com/page.html', 'http://not.example.com/otherpage.html', None), + ('http://example.com/page.html', 'http://www.example.com/otherpage.html', None), # Different port: do NOT send referrer - ('https://example.com:444/page.html', 'https://example.com/not-page.html', None), - ('http://example.com:81/page.html', 'http://example.com/not-page.html', None), - ('http://example.com/page.html', 'http://example.com:81/not-page.html', None), + ('https://example.com:444/page.html', 'https://example.com/not-page.html', None), + ('http://example.com:81/page.html', 'http://example.com/not-page.html', None), + ('http://example.com/page.html', 'http://example.com:81/not-page.html', None), # Different protocols: do NOT send refferer - ('https://example.com/page.html', 'http://example.com/not-page.html', None), - ('https://example.com/page.html', 'http://not.example.com/', None), - ('ftps://example.com/urls.zip', 'https://example.com/not-page.html', None), - ('ftp://example.com/urls.zip', 'http://example.com/not-page.html', None), - ('ftps://example.com/urls.zip', 'https://example.com/not-page.html', None), + ('https://example.com/page.html', 'http://example.com/not-page.html', None), + ('https://example.com/page.html', 'http://not.example.com/', None), + ('ftps://example.com/urls.zip', 'https://example.com/not-page.html', None), + ('ftp://example.com/urls.zip', 'http://example.com/not-page.html', None), + ('ftps://example.com/urls.zip', 'https://example.com/not-page.html', None), # test for user/password stripping - ('https://user:password@example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('https://user:password@example.com/page.html', 'http://example.com/not-page.html', None), + ('https://user:password@example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('https://user:password@example.com/page.html', 'http://example.com/not-page.html', None), ] class MixinOrigin: scenarii = [ # TLS or non-TLS to TLS or non-TLS: referrer origin is sent (yes, even for downgrades) - ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/'), - ('https://example.com/page.html', 'https://scrapy.org', b'https://example.com/'), - ('https://example.com/page.html', 'http://scrapy.org', b'https://example.com/'), - ('http://example.com/page.html', 'http://scrapy.org', b'http://example.com/'), + ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/'), + ('https://example.com/page.html', 'https://scrapy.org', b'https://example.com/'), + ('https://example.com/page.html', 'http://scrapy.org', b'https://example.com/'), + ('http://example.com/page.html', 'http://scrapy.org', b'http://example.com/'), # test for user/password stripping ('https://user:password@example.com/page.html', 'http://scrapy.org', b'https://example.com/'), @@ -160,129 +160,129 @@ class MixinOrigin: class MixinStrictOrigin: scenarii = [ # TLS or non-TLS to TLS or non-TLS: referrer origin is sent but not for downgrades - ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/'), - ('https://example.com/page.html', 'https://scrapy.org', b'https://example.com/'), - ('http://example.com/page.html', 'http://scrapy.org', b'http://example.com/'), + ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/'), + ('https://example.com/page.html', 'https://scrapy.org', b'https://example.com/'), + ('http://example.com/page.html', 'http://scrapy.org', b'http://example.com/'), # downgrade: send nothing - ('https://example.com/page.html', 'http://scrapy.org', None), + ('https://example.com/page.html', 'http://scrapy.org', None), # upgrade: send origin - ('http://example.com/page.html', 'https://scrapy.org', b'http://example.com/'), + ('http://example.com/page.html', 'https://scrapy.org', b'http://example.com/'), # test for user/password stripping - ('https://user:password@example.com/page.html', 'https://scrapy.org', b'https://example.com/'), - ('https://user:password@example.com/page.html', 'http://scrapy.org', None), + ('https://user:password@example.com/page.html', 'https://scrapy.org', b'https://example.com/'), + ('https://user:password@example.com/page.html', 'http://scrapy.org', None), ] class MixinOriginWhenCrossOrigin: scenarii = [ # Same origin (protocol, host, port): send referrer - ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('http://example.com/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), - ('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), - ('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'), - ('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'), + ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('http://example.com/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), + ('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), + ('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'), + ('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'), # Different host: send origin as referrer - ('https://example2.com/page.html', 'https://scrapy.org/otherpage.html', b'https://example2.com/'), - ('https://example2.com/page.html', 'https://not.example2.com/otherpage.html', b'https://example2.com/'), - ('http://example2.com/page.html', 'http://not.example2.com/otherpage.html', b'http://example2.com/'), + ('https://example2.com/page.html', 'https://scrapy.org/otherpage.html', b'https://example2.com/'), + ('https://example2.com/page.html', 'https://not.example2.com/otherpage.html', b'https://example2.com/'), + ('http://example2.com/page.html', 'http://not.example2.com/otherpage.html', b'http://example2.com/'), # exact match required - ('http://example2.com/page.html', 'http://www.example2.com/otherpage.html', b'http://example2.com/'), + ('http://example2.com/page.html', 'http://www.example2.com/otherpage.html', b'http://example2.com/'), # Different port: send origin as referrer - ('https://example3.com:444/page.html', 'https://example3.com/not-page.html', b'https://example3.com:444/'), - ('http://example3.com:81/page.html', 'http://example3.com/not-page.html', b'http://example3.com:81/'), + ('https://example3.com:444/page.html', 'https://example3.com/not-page.html', b'https://example3.com:444/'), + ('http://example3.com:81/page.html', 'http://example3.com/not-page.html', b'http://example3.com:81/'), # Different protocols: send origin as referrer - ('https://example4.com/page.html', 'http://example4.com/not-page.html', b'https://example4.com/'), - ('https://example4.com/page.html', 'http://not.example4.com/', b'https://example4.com/'), - ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), - ('ftp://example4.com/urls.zip', 'http://example4.com/not-page.html', b'ftp://example4.com/'), - ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), + ('https://example4.com/page.html', 'http://example4.com/not-page.html', b'https://example4.com/'), + ('https://example4.com/page.html', 'http://not.example4.com/', b'https://example4.com/'), + ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), + ('ftp://example4.com/urls.zip', 'http://example4.com/not-page.html', b'ftp://example4.com/'), + ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), # test for user/password stripping - ('https://user:password@example5.com/page.html', 'https://example5.com/not-page.html', b'https://example5.com/page.html'), + ('https://user:password@example5.com/page.html', 'https://example5.com/not-page.html', b'https://example5.com/page.html'), # TLS to non-TLS downgrade: send origin - ('https://user:password@example5.com/page.html', 'http://example5.com/not-page.html', b'https://example5.com/'), + ('https://user:password@example5.com/page.html', 'http://example5.com/not-page.html', b'https://example5.com/'), ] class MixinStrictOriginWhenCrossOrigin: scenarii = [ # Same origin (protocol, host, port): send referrer - ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('http://example.com/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), - ('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), - ('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), - ('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'), - ('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'), + ('https://example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('http://example.com/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), + ('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'), + ('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'), + ('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'), + ('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'), # Different host: send origin as referrer - ('https://example2.com/page.html', 'https://scrapy.org/otherpage.html', b'https://example2.com/'), - ('https://example2.com/page.html', 'https://not.example2.com/otherpage.html', b'https://example2.com/'), - ('http://example2.com/page.html', 'http://not.example2.com/otherpage.html', b'http://example2.com/'), + ('https://example2.com/page.html', 'https://scrapy.org/otherpage.html', b'https://example2.com/'), + ('https://example2.com/page.html', 'https://not.example2.com/otherpage.html', b'https://example2.com/'), + ('http://example2.com/page.html', 'http://not.example2.com/otherpage.html', b'http://example2.com/'), # exact match required - ('http://example2.com/page.html', 'http://www.example2.com/otherpage.html', b'http://example2.com/'), + ('http://example2.com/page.html', 'http://www.example2.com/otherpage.html', b'http://example2.com/'), # Different port: send origin as referrer - ('https://example3.com:444/page.html', 'https://example3.com/not-page.html', b'https://example3.com:444/'), - ('http://example3.com:81/page.html', 'http://example3.com/not-page.html', b'http://example3.com:81/'), + ('https://example3.com:444/page.html', 'https://example3.com/not-page.html', b'https://example3.com:444/'), + ('http://example3.com:81/page.html', 'http://example3.com/not-page.html', b'http://example3.com:81/'), # downgrade - ('https://example4.com/page.html', 'http://example4.com/not-page.html', None), - ('https://example4.com/page.html', 'http://not.example4.com/', None), + ('https://example4.com/page.html', 'http://example4.com/not-page.html', None), + ('https://example4.com/page.html', 'http://not.example4.com/', None), # non-TLS to non-TLS - ('ftp://example4.com/urls.zip', 'http://example4.com/not-page.html', b'ftp://example4.com/'), + ('ftp://example4.com/urls.zip', 'http://example4.com/not-page.html', b'ftp://example4.com/'), # upgrade - ('http://example4.com/page.html', 'https://example4.com/not-page.html', b'http://example4.com/'), - ('http://example4.com/page.html', 'https://not.example4.com/', b'http://example4.com/'), + ('http://example4.com/page.html', 'https://example4.com/not-page.html', b'http://example4.com/'), + ('http://example4.com/page.html', 'https://not.example4.com/', b'http://example4.com/'), # Different protocols: send origin as referrer - ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), - ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), + ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), + ('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'), # test for user/password stripping - ('https://user:password@example5.com/page.html', 'https://example5.com/not-page.html', b'https://example5.com/page.html'), + ('https://user:password@example5.com/page.html', 'https://example5.com/not-page.html', b'https://example5.com/page.html'), # TLS to non-TLS downgrade: send nothing - ('https://user:password@example5.com/page.html', 'http://example5.com/not-page.html', None), + ('https://user:password@example5.com/page.html', 'http://example5.com/not-page.html', None), ] class MixinUnsafeUrl: scenarii = [ # TLS to TLS: send referrer - ('https://example.com/sekrit.html', 'http://not.example.com/', b'https://example.com/sekrit.html'), - ('https://example1.com/page.html', 'https://not.example1.com/', b'https://example1.com/page.html'), - ('https://example1.com/page.html', 'https://scrapy.org/', b'https://example1.com/page.html'), - ('https://example1.com:443/page.html', 'https://scrapy.org/', b'https://example1.com/page.html'), - ('https://example1.com:444/page.html', 'https://scrapy.org/', b'https://example1.com:444/page.html'), - ('ftps://example1.com/urls.zip', 'https://scrapy.org/', b'ftps://example1.com/urls.zip'), + ('https://example.com/sekrit.html', 'http://not.example.com/', b'https://example.com/sekrit.html'), + ('https://example1.com/page.html', 'https://not.example1.com/', b'https://example1.com/page.html'), + ('https://example1.com/page.html', 'https://scrapy.org/', b'https://example1.com/page.html'), + ('https://example1.com:443/page.html', 'https://scrapy.org/', b'https://example1.com/page.html'), + ('https://example1.com:444/page.html', 'https://scrapy.org/', b'https://example1.com:444/page.html'), + ('ftps://example1.com/urls.zip', 'https://scrapy.org/', b'ftps://example1.com/urls.zip'), # TLS to non-TLS: send referrer (yes, it's unsafe) - ('https://example2.com/page.html', 'http://not.example2.com/', b'https://example2.com/page.html'), - ('https://example2.com/page.html', 'http://scrapy.org/', b'https://example2.com/page.html'), - ('ftps://example2.com/urls.zip', 'http://scrapy.org/', b'ftps://example2.com/urls.zip'), + ('https://example2.com/page.html', 'http://not.example2.com/', b'https://example2.com/page.html'), + ('https://example2.com/page.html', 'http://scrapy.org/', b'https://example2.com/page.html'), + ('ftps://example2.com/urls.zip', 'http://scrapy.org/', b'ftps://example2.com/urls.zip'), # non-TLS to TLS or non-TLS: send referrer (yes, it's unsafe) - ('http://example3.com/page.html', 'https://not.example3.com/', b'http://example3.com/page.html'), - ('http://example3.com/page.html', 'https://scrapy.org/', b'http://example3.com/page.html'), - ('http://example3.com:8080/page.html', 'https://scrapy.org/', b'http://example3.com:8080/page.html'), - ('http://example3.com:80/page.html', 'http://not.example3.com/', b'http://example3.com/page.html'), - ('http://example3.com/page.html', 'http://scrapy.org/', b'http://example3.com/page.html'), - ('http://example3.com:443/page.html', 'http://scrapy.org/', b'http://example3.com:443/page.html'), - ('ftp://example3.com/urls.zip', 'http://scrapy.org/', b'ftp://example3.com/urls.zip'), - ('ftp://example3.com/urls.zip', 'https://scrapy.org/', b'ftp://example3.com/urls.zip'), + ('http://example3.com/page.html', 'https://not.example3.com/', b'http://example3.com/page.html'), + ('http://example3.com/page.html', 'https://scrapy.org/', b'http://example3.com/page.html'), + ('http://example3.com:8080/page.html', 'https://scrapy.org/', b'http://example3.com:8080/page.html'), + ('http://example3.com:80/page.html', 'http://not.example3.com/', b'http://example3.com/page.html'), + ('http://example3.com/page.html', 'http://scrapy.org/', b'http://example3.com/page.html'), + ('http://example3.com:443/page.html', 'http://scrapy.org/', b'http://example3.com:443/page.html'), + ('ftp://example3.com/urls.zip', 'http://scrapy.org/', b'ftp://example3.com/urls.zip'), + ('ftp://example3.com/urls.zip', 'https://scrapy.org/', b'ftp://example3.com/urls.zip'), # test for user/password stripping - ('http://user:password@example4.com/page.html', 'https://not.example4.com/', b'http://example4.com/page.html'), - ('https://user:password@example4.com/page.html', 'http://scrapy.org/', b'https://example4.com/page.html'), + ('http://user:password@example4.com/page.html', 'https://not.example4.com/', b'http://example4.com/page.html'), + ('https://user:password@example4.com/page.html', 'http://scrapy.org/', b'https://example4.com/page.html'), ] @@ -339,12 +339,12 @@ class CustomPythonOrgPolicy(ReferrerPolicy): class TestSettingsCustomPolicy(TestRefererMiddleware): settings = {'REFERRER_POLICY': 'tests.test_spidermiddleware_referer.CustomPythonOrgPolicy'} scenarii = [ - ('https://example.com/', 'https://scrapy.org/', b'https://python.org/'), - ('http://example.com/', 'http://scrapy.org/', b'http://python.org/'), - ('http://example.com/', 'https://scrapy.org/', b'https://python.org/'), - ('https://example.com/', 'http://scrapy.org/', b'http://python.org/'), - ('file:///home/path/to/somefile.html', 'https://scrapy.org/', b'https://python.org/'), - ('file:///home/path/to/somefile.html', 'http://scrapy.org/', b'http://python.org/'), + ('https://example.com/', 'https://scrapy.org/', b'https://python.org/'), + ('http://example.com/', 'http://scrapy.org/', b'http://python.org/'), + ('http://example.com/', 'https://scrapy.org/', b'https://python.org/'), + ('https://example.com/', 'http://scrapy.org/', b'http://python.org/'), + ('file:///home/path/to/somefile.html', 'https://scrapy.org/', b'https://python.org/'), + ('file:///home/path/to/somefile.html', 'http://scrapy.org/', b'http://python.org/'), ] @@ -541,7 +541,8 @@ class TestReferrerOnRedirect(TestRefererMiddleware): settings = {'REFERRER_POLICY': 'scrapy.spidermiddlewares.referer.UnsafeUrlPolicy'} scenarii = [ - ( 'http://scrapytest.org/1', # parent + ( + 'http://scrapytest.org/1', # parent 'http://scrapytest.org/2', # target ( # redirections: code, URL @@ -551,7 +552,8 @@ class TestReferrerOnRedirect(TestRefererMiddleware): b'http://scrapytest.org/1', # expected initial referer b'http://scrapytest.org/1', # expected referer for the redirection request ), - ( 'https://scrapytest.org/1', + ( + 'https://scrapytest.org/1', 'https://scrapytest.org/2', ( # redirecting to non-secure URL @@ -560,7 +562,8 @@ class TestReferrerOnRedirect(TestRefererMiddleware): b'https://scrapytest.org/1', b'https://scrapytest.org/1', ), - ( 'https://scrapytest.org/1', + ( + 'https://scrapytest.org/1', 'https://scrapytest.com/2', ( # redirecting to non-secure URL: different origin @@ -602,7 +605,8 @@ class TestReferrerOnRedirectNoReferrer(TestReferrerOnRedirect): """ settings = {'REFERRER_POLICY': 'no-referrer'} scenarii = [ - ( 'http://scrapytest.org/1', # parent + ( + 'http://scrapytest.org/1', # parent 'http://scrapytest.org/2', # target ( # redirections: code, URL @@ -612,7 +616,8 @@ class TestReferrerOnRedirectNoReferrer(TestReferrerOnRedirect): None, # expected initial "Referer" None, # expected "Referer" for the redirection request ), - ( 'https://scrapytest.org/1', + ( + 'https://scrapytest.org/1', 'https://scrapytest.org/2', ( (301, 'http://scrapytest.org/3'), @@ -620,7 +625,8 @@ class TestReferrerOnRedirectNoReferrer(TestReferrerOnRedirect): None, None, ), - ( 'https://scrapytest.org/1', + ( + 'https://scrapytest.org/1', 'https://example.com/2', # different origin ( (301, 'http://scrapytest.com/3'), @@ -641,7 +647,8 @@ class TestReferrerOnRedirectSameOrigin(TestReferrerOnRedirect): """ settings = {'REFERRER_POLICY': 'same-origin'} scenarii = [ - ( 'http://scrapytest.org/101', # origin + ( + 'http://scrapytest.org/101', # origin 'http://scrapytest.org/102', # target ( # redirections: code, URL @@ -651,7 +658,8 @@ class TestReferrerOnRedirectSameOrigin(TestReferrerOnRedirect): b'http://scrapytest.org/101', # expected initial "Referer" b'http://scrapytest.org/101', # expected referer for the redirection request ), - ( 'https://scrapytest.org/201', + ( + 'https://scrapytest.org/201', 'https://scrapytest.org/202', ( # redirecting from secure to non-secure URL == different origin @@ -660,7 +668,8 @@ class TestReferrerOnRedirectSameOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/201', None, ), - ( 'https://scrapytest.org/301', + ( + 'https://scrapytest.org/301', 'https://scrapytest.org/302', ( # different domain == different origin @@ -683,7 +692,8 @@ class TestReferrerOnRedirectStrictOrigin(TestReferrerOnRedirect): """ settings = {'REFERRER_POLICY': POLICY_STRICT_ORIGIN} scenarii = [ - ( 'http://scrapytest.org/101', + ( + 'http://scrapytest.org/101', 'http://scrapytest.org/102', ( (301, 'http://scrapytest.org/103'), @@ -692,7 +702,8 @@ class TestReferrerOnRedirectStrictOrigin(TestReferrerOnRedirect): b'http://scrapytest.org/', # send origin b'http://scrapytest.org/', # redirects to same origin: send origin ), - ( 'https://scrapytest.org/201', + ( + 'https://scrapytest.org/201', 'https://scrapytest.org/202', ( # redirecting to non-secure URL: no referrer @@ -701,7 +712,8 @@ class TestReferrerOnRedirectStrictOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/', None, ), - ( 'https://scrapytest.org/301', + ( + 'https://scrapytest.org/301', 'https://scrapytest.org/302', ( # redirecting to non-secure URL (different domain): no referrer @@ -710,7 +722,8 @@ class TestReferrerOnRedirectStrictOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/', None, ), - ( 'http://scrapy.org/401', + ( + 'http://scrapy.org/401', 'http://example.com/402', ( (301, 'http://scrapytest.org/403'), @@ -718,7 +731,8 @@ class TestReferrerOnRedirectStrictOrigin(TestReferrerOnRedirect): b'http://scrapy.org/', b'http://scrapy.org/', ), - ( 'https://scrapy.org/501', + ( + 'https://scrapy.org/501', 'https://example.com/502', ( # HTTPS all along, so origin referrer is kept as-is @@ -728,7 +742,8 @@ class TestReferrerOnRedirectStrictOrigin(TestReferrerOnRedirect): b'https://scrapy.org/', b'https://scrapy.org/', ), - ( 'https://scrapytest.org/601', + ( + 'https://scrapytest.org/601', 'http://scrapytest.org/602', # TLS to non-TLS: no referrer ( (301, 'https://scrapytest.org/603'), # TLS URL again: (still) no referrer @@ -750,7 +765,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): """ settings = {'REFERRER_POLICY': POLICY_ORIGIN_WHEN_CROSS_ORIGIN} scenarii = [ - ( 'http://scrapytest.org/101', # origin + ( + 'http://scrapytest.org/101', # origin 'http://scrapytest.org/102', # target + redirection ( # redirections: code, URL @@ -760,7 +776,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): b'http://scrapytest.org/101', # expected initial referer b'http://scrapytest.org/101', # expected referer for the redirection request ), - ( 'https://scrapytest.org/201', + ( + 'https://scrapytest.org/201', 'https://scrapytest.org/202', ( # redirecting to non-secure URL: send origin @@ -769,7 +786,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/201', b'https://scrapytest.org/', ), - ( 'https://scrapytest.org/301', + ( + 'https://scrapytest.org/301', 'https://scrapytest.org/302', ( # redirecting to non-secure URL (different domain): send origin @@ -778,7 +796,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/301', b'https://scrapytest.org/', ), - ( 'http://scrapy.org/401', + ( + 'http://scrapy.org/401', 'http://example.com/402', ( (301, 'http://scrapytest.org/403'), @@ -786,7 +805,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): b'http://scrapy.org/', b'http://scrapy.org/', ), - ( 'https://scrapy.org/501', + ( + 'https://scrapy.org/501', 'https://example.com/502', ( # all different domains: send origin @@ -796,7 +816,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect): b'https://scrapy.org/', b'https://scrapy.org/', ), - ( 'https://scrapytest.org/301', + ( + 'https://scrapytest.org/301', 'http://scrapytest.org/302', # TLS to non-TLS: send origin ( (301, 'https://scrapytest.org/303'), # TLS URL again: send origin (also) @@ -820,7 +841,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): """ settings = {'REFERRER_POLICY': POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN} scenarii = [ - ( 'http://scrapytest.org/101', # origin + ( + 'http://scrapytest.org/101', # origin 'http://scrapytest.org/102', # target + redirection ( # redirections: code, URL @@ -830,7 +852,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): b'http://scrapytest.org/101', # expected initial referer b'http://scrapytest.org/101', # expected referer for the redirection request ), - ( 'https://scrapytest.org/201', + ( + 'https://scrapytest.org/201', 'https://scrapytest.org/202', ( # redirecting to non-secure URL: do not send the "Referer" header @@ -839,7 +862,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/201', None, ), - ( 'https://scrapytest.org/301', + ( + 'https://scrapytest.org/301', 'https://scrapytest.org/302', ( # redirecting to non-secure URL (different domain): send origin @@ -848,7 +872,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): b'https://scrapytest.org/301', None, ), - ( 'http://scrapy.org/401', + ( + 'http://scrapy.org/401', 'http://example.com/402', ( (301, 'http://scrapytest.org/403'), @@ -856,7 +881,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): b'http://scrapy.org/', b'http://scrapy.org/', ), - ( 'https://scrapy.org/501', + ( + 'https://scrapy.org/501', 'https://example.com/502', ( # all different domains: send origin @@ -866,7 +892,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect): b'https://scrapy.org/', b'https://scrapy.org/', ), - ( 'https://scrapytest.org/601', + ( + 'https://scrapytest.org/601', 'http://scrapytest.org/602', # TLS to non-TLS: do not send "Referer" ( (301, 'https://scrapytest.org/603'), # TLS URL again: (still) send nothing diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index 33fc4d570..ec8311298 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -250,10 +250,10 @@ class UtilsCsvTestCase(unittest.TestCase): result = [row for row in csv] self.assertEqual(result, - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) # explicit type check cuz' we no like stinkin' autocasting! yarrr for result_row in result: @@ -266,10 +266,10 @@ class UtilsCsvTestCase(unittest.TestCase): csv = csviter(response, delimiter='\t') self.assertEqual([row for row in csv], - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) def test_csviter_quotechar(self): body1 = get_testdata('feeds', 'feed-sample6.csv') @@ -279,19 +279,19 @@ class UtilsCsvTestCase(unittest.TestCase): csv1 = csviter(response1, quotechar="'") self.assertEqual([row for row in csv1], - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) response2 = TextResponse(url="http://example.com/", body=body2) csv2 = csviter(response2, delimiter="|", quotechar="'") self.assertEqual([row for row in csv2], - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) def test_csviter_wrong_quotechar(self): body = get_testdata('feeds', 'feed-sample6.csv') @@ -299,10 +299,10 @@ class UtilsCsvTestCase(unittest.TestCase): csv = csviter(response) self.assertEqual([row for row in csv], - [{u"'id'": u"1", u"'name'": u"'alpha'", u"'value'": u"'foobar'"}, - {u"'id'": u"2", u"'name'": u"'unicode'", u"'value'": u"'\xfan\xedc\xf3d\xe9\u203d'"}, - {u"'id'": u"'3'", u"'name'": u"'multi'", u"'value'": u"'foo"}, - {u"'id'": u"4", u"'name'": u"'empty'", u"'value'": u""}]) + [{u"'id'": u"1", u"'name'": u"'alpha'", u"'value'": u"'foobar'"}, + {u"'id'": u"2", u"'name'": u"'unicode'", u"'value'": u"'\xfan\xedc\xf3d\xe9\u203d'"}, + {u"'id'": u"'3'", u"'name'": u"'multi'", u"'value'": u"'foo"}, + {u"'id'": u"4", u"'name'": u"'empty'", u"'value'": u""}]) def test_csviter_delimiter_binary_response_assume_utf8_encoding(self): body = get_testdata('feeds', 'feed-sample3.csv').replace(b',', b'\t') @@ -310,10 +310,10 @@ class UtilsCsvTestCase(unittest.TestCase): csv = csviter(response, delimiter='\t') self.assertEqual([row for row in csv], - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) def test_csviter_headers(self): sample = get_testdata('feeds', 'feed-sample3.csv').splitlines() @@ -323,10 +323,10 @@ class UtilsCsvTestCase(unittest.TestCase): csv = csviter(response, headers=[h.decode('utf-8') for h in headers]) self.assertEqual([row for row in csv], - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': u'foo\nbar'}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': u'foo\nbar'}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) def test_csviter_falserow(self): body = get_testdata('feeds', 'feed-sample3.csv') @@ -336,10 +336,10 @@ class UtilsCsvTestCase(unittest.TestCase): csv = csviter(response) self.assertEqual([row for row in csv], - [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, + [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, {u'id': u'2', u'name': u'unicode', u'value': u'\xfan\xedc\xf3d\xe9\u203d'}, - {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, - {u'id': u'4', u'name': u'empty', u'value': u''}]) + {u'id': u'3', u'name': u'multi', u'value': FOOBAR_NL}, + {u'id': u'4', u'name': u'empty', u'value': u''}]) def test_csviter_exception(self): body = get_testdata('feeds', 'feed-sample3.csv') diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 7abff8281..72a16e9b1 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -203,29 +203,29 @@ def create_skipped_scheme_t(args): for k, args in enumerate([ - ('/index', 'file://'), - ('/index.html', 'file://'), - ('./index.html', 'file://'), - ('../index.html', 'file://'), - ('../../index.html', 'file://'), - ('./data/index.html', 'file://'), - ('.hidden/data/index.html', 'file://'), - ('/home/user/www/index.html', 'file://'), - ('//home/user/www/index.html', 'file://'), - ('file:///home/user/www/index.html', 'file://'), + ('/index', 'file://'), + ('/index.html', 'file://'), + ('./index.html', 'file://'), + ('../index.html', 'file://'), + ('../../index.html', 'file://'), + ('./data/index.html', 'file://'), + ('.hidden/data/index.html', 'file://'), + ('/home/user/www/index.html', 'file://'), + ('//home/user/www/index.html', 'file://'), + ('file:///home/user/www/index.html', 'file://'), - ('index.html', 'http://'), - ('example.com', 'http://'), - ('www.example.com', 'http://'), - ('www.example.com/index.html', 'http://'), - ('http://example.com', 'http://'), - ('http://example.com/index.html', 'http://'), - ('localhost', 'http://'), - ('localhost/index.html', 'http://'), + ('index.html', 'http://'), + ('example.com', 'http://'), + ('www.example.com', 'http://'), + ('www.example.com/index.html', 'http://'), + ('http://example.com', 'http://'), + ('http://example.com/index.html', 'http://'), + ('localhost', 'http://'), + ('localhost/index.html', 'http://'), # some corner cases (default to http://) - ('/', 'http://'), - ('.../test', 'http://'), + ('/', 'http://'), + ('.../test', 'http://'), ], start=1): t_method = create_guess_scheme_t(args) diff --git a/tests/test_webclient.py b/tests/test_webclient.py index 6253d5c3f..d4abebbfb 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -53,28 +53,28 @@ class ParseUrlTestCase(unittest.TestCase): def testParse(self): lip = '127.0.0.1' tests = ( - ("http://127.0.0.1?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), - ("http://127.0.0.1/?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), - ("http://127.0.0.1/foo?c=v&c2=v2#frag", ('http', lip, lip, 80, '/foo?c=v&c2=v2')), + ("http://127.0.0.1?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), + ("http://127.0.0.1/?c=v&c2=v2#fragment", ('http', lip, lip, 80, '/?c=v&c2=v2')), + ("http://127.0.0.1/foo?c=v&c2=v2#frag", ('http', lip, lip, 80, '/foo?c=v&c2=v2')), ("http://127.0.0.1:100?c=v&c2=v2#fragment", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), - ("http://127.0.0.1:100/?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), + ("http://127.0.0.1:100/?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/?c=v&c2=v2')), ("http://127.0.0.1:100/foo?c=v&c2=v2#frag", ('http', lip + ':100', lip, 100, '/foo?c=v&c2=v2')), - ("http://127.0.0.1", ('http', lip, lip, 80, '/')), - ("http://127.0.0.1/", ('http', lip, lip, 80, '/')), - ("http://127.0.0.1/foo", ('http', lip, lip, 80, '/foo')), - ("http://127.0.0.1?param=value", ('http', lip, lip, 80, '/?param=value')), + ("http://127.0.0.1", ('http', lip, lip, 80, '/')), + ("http://127.0.0.1/", ('http', lip, lip, 80, '/')), + ("http://127.0.0.1/foo", ('http', lip, lip, 80, '/foo')), + ("http://127.0.0.1?param=value", ('http', lip, lip, 80, '/?param=value')), ("http://127.0.0.1/?param=value", ('http', lip, lip, 80, '/?param=value')), - ("http://127.0.0.1:12345/foo", ('http', lip + ':12345', lip, 12345, '/foo')), - ("http://spam:12345/foo", ('http', 'spam:12345', 'spam', 12345, '/foo')), - ("http://spam.test.org/foo", ('http', 'spam.test.org', 'spam.test.org', 80, '/foo')), + ("http://127.0.0.1:12345/foo", ('http', lip + ':12345', lip, 12345, '/foo')), + ("http://spam:12345/foo", ('http', 'spam:12345', 'spam', 12345, '/foo')), + ("http://spam.test.org/foo", ('http', 'spam.test.org', 'spam.test.org', 80, '/foo')), - ("https://127.0.0.1/foo", ('https', lip, lip, 443, '/foo')), + ("https://127.0.0.1/foo", ('https', lip, lip, 443, '/foo')), ("https://127.0.0.1/?param=value", ('https', lip, lip, 443, '/?param=value')), - ("https://127.0.0.1:12345/", ('https', lip + ':12345', lip, 12345, '/')), + ("https://127.0.0.1:12345/", ('https', lip + ':12345', lip, 12345, '/')), - ("http://scrapytest.org/foo ", ('http', 'scrapytest.org', 'scrapytest.org', 80, '/foo')), - ("http://egg:7890 ", ('http', 'egg:7890', 'egg', 7890, '/')), + ("http://scrapytest.org/foo ", ('http', 'scrapytest.org', 'scrapytest.org', 80, '/foo')), + ("http://egg:7890 ", ('http', 'egg:7890', 'egg', 7890, '/')), ) for url, test in tests: From c3257dc610ccdd963fae8dda330fa337deb53054 Mon Sep 17 00:00:00 2001 From: santoshkosgi Date: Wed, 15 Apr 2020 17:54:10 +0530 Subject: [PATCH 288/313] Change Content-type to Content-Type (#4481) Co-authored-by: santosh --- scrapy/responsetypes.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index ad89d9d22..7c5eeac21 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -71,7 +71,7 @@ class ResponseTypes: cls = Response if b'Content-Type' in headers: cls = self.from_content_type( - content_type=headers[b'Content-type'], + content_type=headers[b'Content-Type'], content_encoding=headers.get(b'Content-Encoding') ) if cls is Response and b'Content-Disposition' in headers: From ac869181fb9118eb3e2dd0cb938cb1c8271bf6fc Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 15 Apr 2020 13:42:35 -0300 Subject: [PATCH 289/313] Update docs/topics/downloader-middleware.rst MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-Authored-By: Adrián Chaves --- docs/topics/downloader-middleware.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index cea5e4564..8d3ea51f3 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -829,7 +829,7 @@ REDIRECT_MAX_TIMES Default: ``20`` The maximum number of redirections that will be followed for a single request. -After this maximum the request's response is returned as is. +After this maximum, the request's response is returned as is. MetaRefreshMiddleware --------------------- From 47a992615a046d20c31bfca6f8b65c3194e9fd30 Mon Sep 17 00:00:00 2001 From: Victor Torres Date: Wed, 15 Apr 2020 19:57:34 -0300 Subject: [PATCH 290/313] serialize requests with callback references as spider attribute You could define a spider attribute that references a callback method but if this method has a different name than your spider attribute, the request serializer is not able to find it on the spider class. With this commit we're fixing this behavior as we're searching for callback references in the spider object itself instead of looking for attributes with the same function's name, that could be different. --- scrapy/utils/reqser.py | 12 ++++++++---- tests/test_utils_reqser.py | 40 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 48 insertions(+), 4 deletions(-) diff --git a/scrapy/utils/reqser.py b/scrapy/utils/reqser.py index 749bbc387..78e13ec10 100644 --- a/scrapy/utils/reqser.py +++ b/scrapy/utils/reqser.py @@ -1,6 +1,8 @@ """ Helper functions for serializing (and deserializing) requests. """ +import inspect + from scrapy.http import Request from scrapy.utils.python import to_unicode from scrapy.utils.misc import load_object @@ -90,10 +92,12 @@ def _find_method(obj, func): pass else: if func_self is obj: - name = func.__func__.__name__ - if _is_private_method(name): - return _mangle_private_name(obj, func, name) - return name + members = inspect.getmembers(obj, predicate=inspect.ismethod) + for name, obj_func in members: + if obj_func.__func__ is func.__func__: + if _is_private_method(name): + return _mangle_private_name(obj, func, name) + return name raise ValueError("Function %s is not a method of: %s" % (func, obj)) diff --git a/tests/test_utils_reqser.py b/tests/test_utils_reqser.py index c7572f02c..cf84f8fbd 100644 --- a/tests/test_utils_reqser.py +++ b/tests/test_utils_reqser.py @@ -69,6 +69,26 @@ class RequestSerializationTest(unittest.TestCase): errback=self.spider.handle_error) self._assert_serializes_ok(r, spider=self.spider) + def test_reference_callback_serialization(self): + r = Request("http://www.example.com", + callback=self.spider.parse_item_reference, + errback=self.spider.handle_error_reference) + self._assert_serializes_ok(r, spider=self.spider) + request_dict = request_to_dict(r, self.spider) + self.assertEqual(request_dict['callback'], 'parse_item_reference') + self.assertEqual(request_dict['errback'], 'handle_error_reference') + + def test_private_reference_callback_serialization(self): + r = Request("http://www.example.com", + callback=self.spider._TestSpider__parse_item_reference, + errback=self.spider._TestSpider__handle_error_reference) + self._assert_serializes_ok(r, spider=self.spider) + request_dict = request_to_dict(r, self.spider) + self.assertEqual(request_dict['callback'], + '_TestSpider__parse_item_reference') + self.assertEqual(request_dict['errback'], + '_TestSpider__handle_error_reference') + def test_private_callback_serialization(self): r = Request("http://www.example.com", callback=self.spider._TestSpider__parse_item_private, @@ -131,8 +151,28 @@ class TestSpiderMixin: pass +def parse_item(response): + pass + + +def handle_error(failure): + pass + + +def private_parse_item(response): + pass + + +def private_handle_error(failure): + pass + + class TestSpider(Spider, TestSpiderMixin): name = 'test' + parse_item_reference = parse_item + handle_error_reference = handle_error + __parse_item_reference = private_parse_item + __handle_error_reference = private_handle_error def parse_item(self, response): pass From 901892dab380d54186ae855bf65a20eb2467de04 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 16 Apr 2020 14:48:38 +0200 Subject: [PATCH 291/313] Fix the hoverxref configuration --- docs/conf.py | 2 -- tox.ini | 6 ++++++ 2 files changed, 6 insertions(+), 2 deletions(-) diff --git a/docs/conf.py b/docs/conf.py index 4414ef637..813417bae 100644 --- a/docs/conf.py +++ b/docs/conf.py @@ -295,8 +295,6 @@ intersphinx_mapping = { # ------------------------------------ hoverxref_auto_ref = True -hoverxref_project = "scrapy" -hoverxref_version = release hoverxref_role_types = { "class": "tooltip", "confval": "tooltip", diff --git a/tox.ini b/tox.ini index b1babc7fd..cd118c921 100644 --- a/tox.ini +++ b/tox.ini @@ -74,11 +74,15 @@ deps = changedir = docs deps = -rdocs/requirements.txt +setenv = + READTHEDOCS_PROJECT=scrapy + READTHEDOCS_VERSION=master [testenv:docs] basepython = python3 changedir = {[docs]changedir} deps = {[docs]deps} +setenv = {[docs]setenv} commands = sphinx-build -W -b html . {envtmpdir}/html @@ -86,6 +90,7 @@ commands = basepython = python3 changedir = {[docs]changedir} deps = {[docs]deps} +setenv = {[docs]setenv} commands = sphinx-build -b coverage . {envtmpdir}/coverage @@ -93,6 +98,7 @@ commands = basepython = python3 changedir = {[docs]changedir} deps = {[docs]deps} +setenv = {[docs]setenv} commands = sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck From e0921cab667a1fefe9730363ed45091edf1250e8 Mon Sep 17 00:00:00 2001 From: Victor Torres Date: Thu, 16 Apr 2020 11:18:56 -0300 Subject: [PATCH 292/313] remove not used code This code is not needed anymore because we're getting the already mangled name when matching func with spider attributes. --- scrapy/utils/reqser.py | 16 ---------------- tests/test_utils_reqser.py | 37 +------------------------------------ 2 files changed, 1 insertion(+), 52 deletions(-) diff --git a/scrapy/utils/reqser.py b/scrapy/utils/reqser.py index 78e13ec10..1392b2c61 100644 --- a/scrapy/utils/reqser.py +++ b/scrapy/utils/reqser.py @@ -70,20 +70,6 @@ def request_from_dict(d, spider=None): ) -def _is_private_method(name): - return name.startswith('__') and not name.endswith('__') - - -def _mangle_private_name(obj, func, name): - qualname = getattr(func, '__qualname__', None) - if qualname is None: - classname = obj.__class__.__name__.lstrip('_') - return '_%s%s' % (classname, name) - else: - splits = qualname.split('.') - return '_%s%s' % (splits[-2], splits[-1]) - - def _find_method(obj, func): if obj: try: @@ -95,8 +81,6 @@ def _find_method(obj, func): members = inspect.getmembers(obj, predicate=inspect.ismethod) for name, obj_func in members: if obj_func.__func__ is func.__func__: - if _is_private_method(name): - return _mangle_private_name(obj, func, name) return name raise ValueError("Function %s is not a method of: %s" % (func, obj)) diff --git a/tests/test_utils_reqser.py b/tests/test_utils_reqser.py index cf84f8fbd..47853d812 100644 --- a/tests/test_utils_reqser.py +++ b/tests/test_utils_reqser.py @@ -2,7 +2,7 @@ import unittest from scrapy.http import Request, FormRequest from scrapy.spiders import Spider -from scrapy.utils.reqser import request_to_dict, request_from_dict, _is_private_method, _mangle_private_name +from scrapy.utils.reqser import request_to_dict, request_from_dict class RequestSerializationTest(unittest.TestCase): @@ -101,41 +101,6 @@ class RequestSerializationTest(unittest.TestCase): errback=self.spider.handle_error) self._assert_serializes_ok(r, spider=self.spider) - def test_private_callback_name_matching(self): - self.assertTrue(_is_private_method('__a')) - self.assertTrue(_is_private_method('__a_')) - self.assertTrue(_is_private_method('__a_a')) - self.assertTrue(_is_private_method('__a_a_')) - self.assertTrue(_is_private_method('__a__a')) - self.assertTrue(_is_private_method('__a__a_')) - self.assertTrue(_is_private_method('__a___a')) - self.assertTrue(_is_private_method('__a___a_')) - self.assertTrue(_is_private_method('___a')) - self.assertTrue(_is_private_method('___a_')) - self.assertTrue(_is_private_method('___a_a')) - self.assertTrue(_is_private_method('___a_a_')) - self.assertTrue(_is_private_method('____a_a_')) - - self.assertFalse(_is_private_method('_a')) - self.assertFalse(_is_private_method('_a_')) - self.assertFalse(_is_private_method('__a__')) - self.assertFalse(_is_private_method('__')) - self.assertFalse(_is_private_method('___')) - self.assertFalse(_is_private_method('____')) - - def _assert_mangles_to(self, obj, name): - func = getattr(obj, name) - self.assertEqual( - _mangle_private_name(obj, func, func.__name__), - name - ) - - def test_private_name_mangling(self): - self._assert_mangles_to( - self.spider, '_TestSpider__parse_item_private') - self._assert_mangles_to( - self.spider, '_TestSpiderMixin__mixin_callback') - def test_unserializable_callback1(self): r = Request("http://www.example.com", callback=lambda x: x) self.assertRaises(ValueError, request_to_dict, r) From 94c95020b391c3298f4a7fd7608d48d74117bf43 Mon Sep 17 00:00:00 2001 From: Victor Torres Date: Thu, 16 Apr 2020 11:37:03 -0300 Subject: [PATCH 293/313] add comment to explain the use of __func__ instead of instance method objects --- scrapy/utils/reqser.py | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/scrapy/utils/reqser.py b/scrapy/utils/reqser.py index 1392b2c61..5ea2aafb8 100644 --- a/scrapy/utils/reqser.py +++ b/scrapy/utils/reqser.py @@ -80,6 +80,13 @@ def _find_method(obj, func): if func_self is obj: members = inspect.getmembers(obj, predicate=inspect.ismethod) for name, obj_func in members: + # We need to use __func__ to access the original + # function object because instance method objects + # are generated each time attribute is retrieved from + # instance. + # + # Reference: The standard type hierarchy + # https://docs.python.org/3/reference/datamodel.html if obj_func.__func__ is func.__func__: return name raise ValueError("Function %s is not a method of: %s" % (func, obj)) From c9229922772a4d7f92a26786d6ea441609043a09 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Thu, 16 Apr 2020 11:37:37 -0300 Subject: [PATCH 294/313] Tests: Move code inside __main__ block --- tests/CrawlerRunner/ip_address.py | 21 +++++++++++---------- 1 file changed, 11 insertions(+), 10 deletions(-) diff --git a/tests/CrawlerRunner/ip_address.py b/tests/CrawlerRunner/ip_address.py index 5a71536d8..826374cd4 100644 --- a/tests/CrawlerRunner/ip_address.py +++ b/tests/CrawlerRunner/ip_address.py @@ -23,15 +23,16 @@ class LocalhostSpider(Spider): self.logger.info("IP address: %s" % response.ip_address) -with MockServer() as mock_http_server, MockDNSServer() as mock_dns_server: - port = urlparse(mock_http_server.http_address).port - url = "http://not.a.real.domain:{port}/echo".format(port=port) +if __name__ == "__main__": + with MockServer() as mock_http_server, MockDNSServer() as mock_dns_server: + port = urlparse(mock_http_server.http_address).port + url = "http://not.a.real.domain:{port}/echo".format(port=port) - servers = [(mock_dns_server.host, mock_dns_server.port)] - reactor.installResolver(createResolver(servers=servers)) + servers = [(mock_dns_server.host, mock_dns_server.port)] + reactor.installResolver(createResolver(servers=servers)) - configure_logging() - runner = CrawlerRunner() - d = runner.crawl(LocalhostSpider, url=url) - d.addBoth(lambda _: reactor.stop()) - reactor.run() + configure_logging() + runner = CrawlerRunner() + d = runner.crawl(LocalhostSpider, url=url) + d.addBoth(lambda _: reactor.stop()) + reactor.run() From 1ade3fc723d1e5d7b6a3300b454d32656bdb8d28 Mon Sep 17 00:00:00 2001 From: Victor Torres Date: Fri, 17 Apr 2020 10:34:34 -0300 Subject: [PATCH 295/313] trying to improve test coverage --- tests/test_utils_reqser.py | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/tests/test_utils_reqser.py b/tests/test_utils_reqser.py index 47853d812..50b026d1c 100644 --- a/tests/test_utils_reqser.py +++ b/tests/test_utils_reqser.py @@ -110,6 +110,21 @@ class RequestSerializationTest(unittest.TestCase): r = Request("http://www.example.com", callback=self.spider.parse_item) self.assertRaises(ValueError, request_to_dict, r) + def test_unserializable_callback3(self): + """Parser method is removed or replaced dynamically.""" + + class MySpider(Spider): + + name = 'my_spider' + + def parse(self, response): + pass + + spider = MySpider() + r = Request("http://www.example.com", callback=spider.parse) + setattr(spider, 'parse', None) + self.assertRaises(ValueError, request_to_dict, r, spider=spider) + class TestSpiderMixin: def __mixin_callback(self, response): From 04b6295a69174e81beceb0b1429fa3775949e99d Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Fri, 17 Apr 2020 20:50:17 -0300 Subject: [PATCH 296/313] Docs: replace deprecated FEED_* settings --- docs/topics/practices.rst | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/docs/topics/practices.rst b/docs/topics/practices.rst index e3e8fdc72..cf1de1bd1 100644 --- a/docs/topics/practices.rst +++ b/docs/topics/practices.rst @@ -35,8 +35,9 @@ Here's an example showing how to run a single spider with it. ... process = CrawlerProcess(settings={ - 'FEED_FORMAT': 'json', - 'FEED_URI': 'items.json' + "FEEDS": { + "items.json": {"format": "json"}, + }, }) process.crawl(MySpider) From bfeb2c8c13de0c45af21228f69395a1131913da5 Mon Sep 17 00:00:00 2001 From: sakshamb2113 <44064539+sakshamb2113@users.noreply.github.com> Date: Sat, 18 Apr 2020 20:51:26 +0530 Subject: [PATCH 297/313] Added warning to use double quotes in Windows for scrapy shell in shell.rst (#4450) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit * modified debugging memory leaks with guppy in leaks.rst * modified leaks.rst(issue #4285) * removed guppy from telnet.py * Fix undefined name error * removed hpy key from telnet_vars in telnet.py * updated shell.rst * Update docs/topics/shell.rst Co-Authored-By: Adrián Chaves Co-authored-by: Adrián Chaves --- docs/topics/shell.rst | 11 +++++++++++ 1 file changed, 11 insertions(+) diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index 8f7518b19..0f46f1c87 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -156,6 +156,17 @@ First, we launch the shell:: scrapy shell 'https://scrapy.org' --nolog +.. note:: + + Remember to always enclose URLs in quotes when running the Scrapy shell from + the command line, otherwise URLs containing arguments (i.e. the ``&`` character) + will not work. + + On Windows, use double quotes instead:: + + scrapy shell "https://scrapy.org" --nolog + + Then, the shell fetches the URL (using the Scrapy downloader) and prints the list of available objects and useful shortcuts (you'll notice that these lines all start with the ``[s]`` prefix):: From e4750f2fbdacbeb7a20ae7c6b13bba3fb0f7ad54 Mon Sep 17 00:00:00 2001 From: Aditya Kumar Date: Mon, 20 Apr 2020 21:17:57 +0530 Subject: [PATCH 298/313] async/deferred signal handlers (#4390) * [docs] async/deferred signal handlers * [docs] update deferred signals example * [docs] add subsections for built-in signals * docs(signals): update signal handler example * docs(signals): update signal handler example --- docs/topics/signals.rst | 96 +++++++++++++++++++++++++++++++++-------- 1 file changed, 77 insertions(+), 19 deletions(-) diff --git a/docs/topics/signals.rst b/docs/topics/signals.rst index 2def53848..8661f86a0 100644 --- a/docs/topics/signals.rst +++ b/docs/topics/signals.rst @@ -16,8 +16,7 @@ deliver the arguments that the handler receives. You can connect to signals (or send your own) through the :ref:`topics-api-signals`. -Here is a simple example showing how you can catch signals and perform some action: -:: +Here is a simple example showing how you can catch signals and perform some action:: from scrapy import signals from scrapy import Spider @@ -52,9 +51,45 @@ Deferred signal handlers ======================== Some signals support returning :class:`~twisted.internet.defer.Deferred` -objects from their handlers, see the :ref:`topics-signals-ref` below to know -which ones. +objects from their handlers, allowing you to run asynchronous code that +does not block Scrapy. If a signal handler returns a +:class:`~twisted.internet.defer.Deferred`, Scrapy waits for that +:class:`~twisted.internet.defer.Deferred` to fire. +Let's take an example:: + + class SignalSpider(scrapy.Spider): + name = 'signals' + start_urls = ['http://quotes.toscrape.com/page/1/'] + + @classmethod + def from_crawler(cls, crawler, *args, **kwargs): + spider = super(SignalSpider, cls).from_crawler(crawler, *args, **kwargs) + crawler.signals.connect(spider.item_scraped, signal=signals.item_scraped) + return spider + + def item_scraped(self, item): + # Send the scraped item to the server + d = treq.post( + 'http://example.com/post', + json.dumps(item).encode('ascii'), + headers={b'Content-Type': [b'application/json']} + ) + + # The next item will be scraped only after + # deferred (d) is fired + return d + + def parse(self, response): + for quote in response.css('div.quote'): + yield { + 'text': quote.css('span.text::text').get(), + 'author': quote.css('small.author::text').get(), + 'tags': quote.css('div.tags a.tag::text').getall(), + } + +See the :ref:`topics-signals-ref` below to know which signals support +:class:`~twisted.internet.defer.Deferred`. .. _topics-signals-ref: @@ -66,9 +101,12 @@ Built-in signals reference Here's the list of Scrapy built-in signals and their meaning. -engine_started +Engine signals -------------- +engine_started +~~~~~~~~~~~~~~ + .. signal:: engine_started .. function:: engine_started() @@ -81,7 +119,7 @@ engine_started getting fired before :signal:`spider_opened`. engine_stopped --------------- +~~~~~~~~~~~~~~ .. signal:: engine_stopped .. function:: engine_stopped() @@ -91,9 +129,20 @@ engine_stopped This signal supports returning deferreds from their handlers. -item_scraped +Item signals ------------ +.. note:: + As at max :setting:`CONCURRENT_ITEMS` items are processed in + parallel, many deferreds are fired together using + :class:`~twisted.internet.defer.DeferredList`. Hence the next + batch waits for the :class:`~twisted.internet.defer.DeferredList` + to fire and then runs the respective item signal handler for + the next batch of scraped items. + +item_scraped +~~~~~~~~~~~~ + .. signal:: item_scraped .. function:: item_scraped(item, response, spider) @@ -112,7 +161,7 @@ item_scraped :type response: :class:`~scrapy.http.Response` object item_dropped ------------- +~~~~~~~~~~~~ .. signal:: item_dropped .. function:: item_dropped(item, response, exception, spider) @@ -137,7 +186,7 @@ item_dropped :type exception: :exc:`~scrapy.exceptions.DropItem` exception item_error ------------- +~~~~~~~~~~ .. signal:: item_error .. function:: item_error(item, response, spider, failure) @@ -159,8 +208,11 @@ item_error :param failure: the exception raised :type failure: twisted.python.failure.Failure +Spider signals +-------------- + spider_closed -------------- +~~~~~~~~~~~~~ .. signal:: spider_closed .. function:: spider_closed(spider, reason) @@ -183,7 +235,7 @@ spider_closed :type reason: str spider_opened -------------- +~~~~~~~~~~~~~ .. signal:: spider_opened .. function:: spider_opened(spider) @@ -198,7 +250,7 @@ spider_opened :type spider: :class:`~scrapy.spiders.Spider` object spider_idle ------------ +~~~~~~~~~~~ .. signal:: spider_idle .. function:: spider_idle(spider) @@ -228,7 +280,7 @@ spider_idle due to duplication). spider_error ------------- +~~~~~~~~~~~~ .. signal:: spider_error .. function:: spider_error(failure, response, spider) @@ -246,8 +298,11 @@ spider_error :param spider: the spider which raised the exception :type spider: :class:`~scrapy.spiders.Spider` object +Request signals +--------------- + request_scheduled ------------------ +~~~~~~~~~~~~~~~~~ .. signal:: request_scheduled .. function:: request_scheduled(request, spider) @@ -264,7 +319,7 @@ request_scheduled :type spider: :class:`~scrapy.spiders.Spider` object request_dropped ---------------- +~~~~~~~~~~~~~~~ .. signal:: request_dropped .. function:: request_dropped(request, spider) @@ -281,7 +336,7 @@ request_dropped :type spider: :class:`~scrapy.spiders.Spider` object request_reached_downloader ---------------------------- +~~~~~~~~~~~~~~~~~~~~~~~~~~ .. signal:: request_reached_downloader .. function:: request_reached_downloader(request, spider) @@ -297,7 +352,7 @@ request_reached_downloader :type spider: :class:`~scrapy.spiders.Spider` object request_left_downloader ------------------------ +~~~~~~~~~~~~~~~~~~~~~~~ .. signal:: request_left_downloader .. function:: request_left_downloader(request, spider) @@ -315,8 +370,11 @@ request_left_downloader :param spider: the spider that yielded the request :type spider: :class:`~scrapy.spiders.Spider` object +Response signals +---------------- + response_received ------------------ +~~~~~~~~~~~~~~~~~ .. signal:: response_received .. function:: response_received(response, request, spider) @@ -336,7 +394,7 @@ response_received :type spider: :class:`~scrapy.spiders.Spider` object response_downloaded -------------------- +~~~~~~~~~~~~~~~~~~~ .. signal:: response_downloaded .. function:: response_downloaded(response, request, spider) From efb6f13debf9406a214a9cee3d94d47875d542f5 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <1731933+elacuesta@users.noreply.github.com> Date: Thu, 23 Apr 2020 07:40:10 -0300 Subject: [PATCH 299/313] Remove assertions from production code (#4440) --- scrapy/commands/__init__.py | 3 ++- scrapy/contracts/default.py | 6 +++++- scrapy/core/downloader/middleware.py | 5 +++-- scrapy/core/engine.py | 26 ++++++++++++++++++-------- scrapy/core/scraper.py | 6 +++++- scrapy/crawler.py | 3 ++- scrapy/http/request/__init__.py | 3 ++- scrapy/pipelines/files.py | 6 ++++-- scrapy/utils/iterators.py | 10 ++++++---- scrapy/utils/reactor.py | 3 ++- tests/test_utils_iterators.py | 2 +- 11 files changed, 50 insertions(+), 23 deletions(-) diff --git a/scrapy/commands/__init__.py b/scrapy/commands/__init__.py index a573a03d9..9f8e6986a 100644 --- a/scrapy/commands/__init__.py +++ b/scrapy/commands/__init__.py @@ -23,7 +23,8 @@ class ScrapyCommand: self.settings = None # set in scrapy.cmdline def set_crawler(self, crawler): - assert not hasattr(self, '_crawler'), "crawler already set" + if hasattr(self, '_crawler'): + raise RuntimeError("crawler already set") self._crawler = crawler def syntax(self): diff --git a/scrapy/contracts/default.py b/scrapy/contracts/default.py index 3002fc702..a1b0f8f22 100644 --- a/scrapy/contracts/default.py +++ b/scrapy/contracts/default.py @@ -58,7 +58,11 @@ class ReturnsContract(Contract): def __init__(self, *args, **kwargs): super(ReturnsContract, self).__init__(*args, **kwargs) - assert len(self.args) in [1, 2, 3] + if len(self.args) not in [1, 2, 3]: + raise ValueError( + "Incorrect argument quantity: expected 1, 2 or 3, got %i" + % len(self.args) + ) self.obj_name = self.args[0] or None self.obj_type = self.objects[self.obj_name] diff --git a/scrapy/core/downloader/middleware.py b/scrapy/core/downloader/middleware.py index 5a03dcdf7..4c2eea522 100644 --- a/scrapy/core/downloader/middleware.py +++ b/scrapy/core/downloader/middleware.py @@ -45,8 +45,9 @@ class DownloaderMiddlewareManager(MiddlewareManager): @defer.inlineCallbacks def process_response(response): - assert response is not None, 'Received None in process_response' - if isinstance(response, Request): + if response is None: + raise TypeError("Received None in process_response") + elif isinstance(response, Request): return response for method in self.methods['process_response']: diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 66cf9ad9a..77d71846e 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -73,7 +73,8 @@ class ExecutionEngine: @defer.inlineCallbacks def start(self): """Start the execution engine""" - assert not self.running, "Engine already running" + if self.running: + raise RuntimeError("Engine already running") self.start_time = time() yield self.signals.send_catch_log_deferred(signal=signals.engine_started) self.running = True @@ -82,7 +83,8 @@ class ExecutionEngine: def stop(self): """Stop the execution engine gracefully""" - assert self.running, "Engine not running" + if not self.running: + raise RuntimeError("Engine not running") self.running = False dfd = self._close_all_spiders() return dfd.addBoth(lambda _: self._finish_stopping_engine()) @@ -165,7 +167,11 @@ class ExecutionEngine: return d def _handle_downloader_output(self, response, request, spider): - assert isinstance(response, (Request, Response, Failure)), response + if not isinstance(response, (Request, Response, Failure)): + raise TypeError( + "Incorrect type: expected Request, Response or Failure, got %s: %r" + % (type(response), response) + ) # downloader middleware can return requests (for example, redirects) if isinstance(response, Request): self.crawl(response, spider) @@ -205,8 +211,8 @@ class ExecutionEngine: return not bool(self.slot) def crawl(self, request, spider): - assert spider in self.open_spiders, \ - "Spider %r not opened when crawling: %s" % (spider.name, request) + if spider not in self.open_spiders: + raise RuntimeError("Spider %r not opened when crawling: %s" % (spider.name, request)) self.schedule(request, spider) self.slot.nextcall.schedule() @@ -232,7 +238,11 @@ class ExecutionEngine: slot.add_request(request) def _on_success(response): - assert isinstance(response, (Response, Request)) + if not isinstance(response, (Response, Request)): + raise TypeError( + "Incorrect type: expected Response or Request, got %s: %r" + % (type(response), response) + ) if isinstance(response, Response): response.request = request # tie request to response received logkws = self.logformatter.crawled(request, response, spider) @@ -253,8 +263,8 @@ class ExecutionEngine: @defer.inlineCallbacks def open_spider(self, spider, start_requests=(), close_if_idle=True): - assert self.has_capacity(), "No free spider slot when opening %r" % \ - spider.name + if not self.has_capacity(): + raise RuntimeError("No free spider slot when opening %r" % spider.name) logger.info("Spider opened", extra={'spider': spider}) nextcall = CallLaterOnce(self._next_request, spider) scheduler = self.scheduler_cls.from_crawler(self.crawler) diff --git a/scrapy/core/scraper.py b/scrapy/core/scraper.py index 3e4826216..edbb4dd66 100644 --- a/scrapy/core/scraper.py +++ b/scrapy/core/scraper.py @@ -123,7 +123,11 @@ class Scraper: def _scrape(self, response, request, spider): """Handle the downloaded response or failure through the spider callback/errback""" - assert isinstance(response, (Response, Failure)) + if not isinstance(response, (Response, Failure)): + raise TypeError( + "Incorrect type: expected Response or Failure, got %s: %r" + % (type(response), response) + ) dfd = self._scrape2(response, request, spider) # returns spider's processed output dfd.addErrback(self.handle_spider_error, request, response, spider) diff --git a/scrapy/crawler.py b/scrapy/crawler.py index 20990ea41..6f43771e2 100644 --- a/scrapy/crawler.py +++ b/scrapy/crawler.py @@ -78,7 +78,8 @@ class Crawler: @defer.inlineCallbacks def crawl(self, *args, **kwargs): - assert not self.crawling, "Crawling already taking place" + if self.crawling: + raise RuntimeError("Crawling already taking place") self.crawling = True try: diff --git a/scrapy/http/request/__init__.py b/scrapy/http/request/__init__.py index 0a6637af8..a98ba9960 100644 --- a/scrapy/http/request/__init__.py +++ b/scrapy/http/request/__init__.py @@ -24,7 +24,8 @@ class Request(object_ref): self.method = str(method).upper() self._set_url(url) self._set_body(body) - assert isinstance(priority, int), "Request priority not an integer: %r" % priority + if not isinstance(priority, int): + raise TypeError("Request priority not an integer: %r" % priority) self.priority = priority if callback is not None and not callable(callback): diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index aab645d3d..ae365db5b 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -106,7 +106,8 @@ class S3FilesStore: else: from boto.s3.connection import S3Connection self.S3Connection = S3Connection - assert uri.startswith('s3://') + if not uri.startswith("s3://"): + raise ValueError("Incorrect URI scheme in %s, expected 's3'" % uri) self.bucket, self.prefix = uri[5:].split('/', 1) def stat_file(self, path, info): @@ -266,7 +267,8 @@ class FTPFilesStore: USE_ACTIVE_MODE = None def __init__(self, uri): - assert uri.startswith('ftp://') + if not uri.startswith("ftp://"): + raise ValueError("Incorrect URI scheme in %s, expected 'ftp'" % uri) u = urlparse(uri) self.port = u.port self.host = u.hostname diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index b71419111..5e15bf0c8 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -128,10 +128,12 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): def _body_or_str(obj, unicode=True): expected_types = (Response, str, bytes) - assert isinstance(obj, expected_types), \ - "obj must be %s, not %s" % ( - " or ".join(t.__name__ for t in expected_types), - type(obj).__name__) + if not isinstance(obj, expected_types): + expected_types_str = " or ".join(t.__name__ for t in expected_types) + raise TypeError( + "Object %r must be %s, not %s" + % (obj, expected_types_str, type(obj).__name__) + ) if isinstance(obj, Response): if not unicode: return obj.body diff --git a/scrapy/utils/reactor.py b/scrapy/utils/reactor.py index 5308812d6..3c705f69b 100644 --- a/scrapy/utils/reactor.py +++ b/scrapy/utils/reactor.py @@ -9,7 +9,8 @@ from scrapy.utils.misc import load_object def listen_tcp(portrange, host, factory): """Like reactor.listenTCP but tries different ports in a range.""" from twisted.internet import reactor - assert len(portrange) <= 2, "invalid portrange: %s" % portrange + if len(portrange) > 2: + raise ValueError("invalid portrange: %s" % portrange) if not portrange: return reactor.listenTCP(0, factory, interface=host) if not hasattr(portrange, '__iter__'): diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index ec8311298..a85087619 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -157,7 +157,7 @@ class XmliterTestCase(unittest.TestCase): def test_xmliter_objtype_exception(self): i = self.xmliter(42, 'product') - self.assertRaises(AssertionError, next, i) + self.assertRaises(TypeError, next, i) def test_xmliter_encoding(self): body = b'\n\n Some Turkish Characters \xd6\xc7\xde\xdd\xd0\xdc \xfc\xf0\xfd\xfe\xe7\xf6\n\n\n' From ffe576c4ed192882d1e40fef815f0c1d5354249a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Fri, 24 Apr 2020 11:44:36 +0200 Subject: [PATCH 300/313] Cover Scrapy 2.1 in the release notes (#4499) Co-authored-by: Mikhail Korobov --- docs/news.rst | 147 +++++++++++++++++++++++++++++++ docs/topics/request-response.rst | 5 ++ 2 files changed, 152 insertions(+) diff --git a/docs/news.rst b/docs/news.rst index e9b7140cd..a158246eb 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -3,6 +3,153 @@ Release notes ============= +.. _release-2.1.0: + +Scrapy 2.1.0 (2020-04-24) +------------------------- + +Highlights: + +* New :setting:`FEEDS` setting to export to multiple feeds +* New :attr:`Response.ip_address ` attribute + +Backward-incompatible changes +~~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +* :exc:`AssertionError` exceptions triggered by :ref:`assert ` + statements have been replaced by new exception types, to support running + Python in optimized mode (see :option:`-O`) without changing Scrapy’s + behavior in any unexpected ways. + + If you catch an :exc:`AssertionError` exception from Scrapy, update your + code to catch the corresponding new exception. + + (:issue:`4440`) + + +Deprecation removals +~~~~~~~~~~~~~~~~~~~~ + +* The ``LOG_UNSERIALIZABLE_REQUESTS`` setting is no longer supported, use + :setting:`SCHEDULER_DEBUG` instead (:issue:`4385`) + +* The ``REDIRECT_MAX_METAREFRESH_DELAY`` setting is no longer supported, use + :setting:`METAREFRESH_MAXDELAY` instead (:issue:`4385`) + +* The :class:`~scrapy.downloadermiddlewares.chunked.ChunkedTransferMiddleware` + middleware has been removed, including the entire + :class:`scrapy.downloadermiddlewares.chunked` module; chunked transfers + work out of the box (:issue:`4431`) + +* The ``spiders`` property has been removed from + :class:`~scrapy.crawler.Crawler`, use :class:`CrawlerRunner.spider_loader + ` or instantiate + :setting:`SPIDER_LOADER_CLASS` with your settings instead (:issue:`4398`) + +* The ``MultiValueDict``, ``MultiValueDictKeyError``, and ``SiteNode`` + classes have been removed from :mod:`scrapy.utils.datatypes` + (:issue:`4400`) + + +Deprecations +~~~~~~~~~~~~ + +* The ``FEED_FORMAT`` and ``FEED_URI`` settings have been deprecated in + favor of the new :setting:`FEEDS` setting (:issue:`1336`, :issue:`3858`, + :issue:`4507`) + + +New features +~~~~~~~~~~~~ + +* A new setting, :setting:`FEEDS`, allows configuring multiple output feeds + with different settings each (:issue:`1336`, :issue:`3858`, :issue:`4507`) + +* The :command:`crawl` and :command:`runspider` commands now support multiple + ``-o`` parameters (:issue:`1336`, :issue:`3858`, :issue:`4507`) + +* The :command:`crawl` and :command:`runspider` commands now support + specifying an output format by appending ``:`` to the output file + (:issue:`1336`, :issue:`3858`, :issue:`4507`) + +* The new :attr:`Response.ip_address ` + attribute gives access to the IP address that originated a response + (:issue:`3903`, :issue:`3940`) + +* A warning is now issued when a value in + :attr:`~scrapy.spiders.Spider.allowed_domains` includes a port + (:issue:`50`, :issue:`3198`, :issue:`4413`) + +* Zsh completion now excludes used option aliases from the completion list + (:issue:`4438`) + + +Bug fixes +~~~~~~~~~ + +* :ref:`Request serialization ` no longer breaks for + callbacks that are spider attributes which are assigned a function with a + different name (:issue:`4500`) + +* ``None`` values in :attr:`~scrapy.spiders.Spider.allowed_domains` no longer + cause a :exc:`TypeError` exception (:issue:`4410`) + +* Zsh completion no longer allows options after arguments (:issue:`4438`) + +* zope.interface 5.0.0 and later versions are now supported + (:issue:`4447`, :issue:`4448`) + +* :meth:`Spider.make_requests_from_url + `, deprecated in Scrapy + 1.4.0, now issues a warning when used (:issue:`4412`) + + +Documentation +~~~~~~~~~~~~~ + +* Improved the documentation about signals that allow their handlers to + return a :class:`~twisted.internet.defer.Deferred` (:issue:`4295`, + :issue:`4390`) + +* Our PyPI entry now includes links for our documentation, our source code + repository and our issue tracker (:issue:`4456`) + +* Covered the `curl2scrapy `_ + service in the documentation (:issue:`4206`, :issue:`4455`) + +* Removed references to the Guppy library, which only works in Python 2 + (:issue:`4285`, :issue:`4343`) + +* Extended use of InterSphinx to link to Python 3 documentation + (:issue:`4444`, :issue:`4445`) + +* Added support for Sphinx 3.0 and later (:issue:`4475`, :issue:`4480`, + :issue:`4496`, :issue:`4503`) + + +Quality assurance +~~~~~~~~~~~~~~~~~ + +* Removed warnings about using old, removed settings (:issue:`4404`) + +* Removed a warning about importing + :class:`~twisted.internet.testing.StringTransport` from + ``twisted.test.proto_helpers`` in Twisted 19.7.0 or newer (:issue:`4409`) + +* Removed outdated Debian package build files (:issue:`4384`) + +* Removed :class:`object` usage as a base class (:issue:`4430`) + +* Removed code that added support for old versions of Twisted that we no + longer support (:issue:`4472`) + +* Fixed code style issues (:issue:`4468`, :issue:`4469`, :issue:`4471`, + :issue:`4481`) + +* Removed :func:`twisted.internet.defer.returnValue` calls (:issue:`4443`, + :issue:`4446`, :issue:`4489`) + + .. _release-2.0.1: Scrapy 2.0.1 (2020-03-18) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 5eb4915cd..024f46466 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -619,6 +619,9 @@ Response objects :param ip_address: The IP address of the server from which the Response originated. :type ip_address: :class:`ipaddress.IPv4Address` or :class:`ipaddress.IPv6Address` + .. versionadded:: 2.1.0 + The ``ip_address`` parameter. + .. attribute:: Response.url A string containing the URL of the response. @@ -710,6 +713,8 @@ Response objects .. attribute:: Response.ip_address + .. versionadded:: 2.1.0 + The IP address of the server from which the Response originated. This attribute is currently only populated by the HTTP 1.1 download From 3878b67a3771102d4b6668ac749afbec7dc85a8f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Fri, 24 Apr 2020 11:46:54 +0200 Subject: [PATCH 301/313] =?UTF-8?q?Bump=20version:=202.0.0=20=E2=86=92=202?= =?UTF-8?q?.1.0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.cfg | 2 +- scrapy/VERSION | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/.bumpversion.cfg b/.bumpversion.cfg index f347a0cd0..de22a2783 100644 --- a/.bumpversion.cfg +++ b/.bumpversion.cfg @@ -1,5 +1,5 @@ [bumpversion] -current_version = 2.0.0 +current_version = 2.1.0 commit = True tag = True tag_name = {new_version} diff --git a/scrapy/VERSION b/scrapy/VERSION index 227cea215..7ec1d6db4 100644 --- a/scrapy/VERSION +++ b/scrapy/VERSION @@ -1 +1 @@ -2.0.0 +2.1.0 From c207dbf939811176a7b094e0f2547aa7846b1cf8 Mon Sep 17 00:00:00 2001 From: Ashe Date: Tue, 28 Apr 2020 02:45:19 +0900 Subject: [PATCH 302/313] Remove the asyncio warning from coroutines page (#4513) --- docs/topics/coroutines.rst | 4 ---- 1 file changed, 4 deletions(-) diff --git a/docs/topics/coroutines.rst b/docs/topics/coroutines.rst index 5f61d6796..7a9ecd4d5 100644 --- a/docs/topics/coroutines.rst +++ b/docs/topics/coroutines.rst @@ -7,10 +7,6 @@ Coroutines Scrapy has :ref:`partial support ` for the :ref:`coroutine syntax `. -.. warning:: :mod:`asyncio` support in Scrapy is experimental. Future Scrapy - versions may introduce related API and behavior changes without a - deprecation period or warning. - .. _coroutine-support: Supported callables From e3c3ec2ba988f654be1676586714fd96dba32c23 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 28 Apr 2020 13:48:50 +0200 Subject: [PATCH 303/313] Run quick tests first in Travis CI --- .travis.yml | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/.travis.yml b/.travis.yml index 66e1a9617..dc91dfe4c 100644 --- a/.travis.yml +++ b/.travis.yml @@ -11,6 +11,9 @@ matrix: python: 3.8 - env: TOXENV=flake8 python: 3.8 + - env: TOXENV=docs + python: 3.7 # Keep in sync with .readthedocs.yml + - env: TOXENV=pypy3 - env: TOXENV=py35 python: 3.5 @@ -28,8 +31,6 @@ matrix: python: 3.8 - env: TOXENV=py38-asyncio python: 3.8 - - env: TOXENV=docs - python: 3.7 # Keep in sync with .readthedocs.yml install: - | if [ "$TOXENV" = "pypy3" ]; then From 5c0f11b4ef1d58de4245d9f4ac9a26f21faf082c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 28 Apr 2020 17:32:53 +0200 Subject: [PATCH 304/313] Simplify the asyncio Tox environment --- .travis.yml | 4 ++-- tox.ini | 12 +----------- 2 files changed, 3 insertions(+), 13 deletions(-) diff --git a/.travis.yml b/.travis.yml index 66e1a9617..a924eb68c 100644 --- a/.travis.yml +++ b/.travis.yml @@ -16,7 +16,7 @@ matrix: python: 3.5 - env: TOXENV=pinned python: 3.5 - - env: TOXENV=py35-asyncio + - env: TOXENV=asyncio python: 3.5.2 - env: TOXENV=py36 python: 3.6 @@ -26,7 +26,7 @@ matrix: python: 3.8 - env: TOXENV=extra-deps python: 3.8 - - env: TOXENV=py38-asyncio + - env: TOXENV=asyncio python: 3.8 - env: TOXENV=docs python: 3.7 # Keep in sync with .readthedocs.yml diff --git a/tox.ini b/tox.ini index cd118c921..697328ebd 100644 --- a/tox.ini +++ b/tox.ini @@ -102,16 +102,6 @@ setenv = {[docs]setenv} commands = sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck -[asyncio] +[testenv:asyncio] commands = {[testenv]commands} --reactor=asyncio - -[testenv:py35-asyncio] -basepython = python3.5 -deps = {[testenv]deps} -commands = {[asyncio]commands} - -[testenv:py38-asyncio] -basepython = python3.8 -deps = {[testenv]deps} -commands = {[asyncio]commands} From 3a64f3eb2902ed8168b78c43f3516cf657873cef Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 28 Apr 2020 17:44:19 +0200 Subject: [PATCH 305/313] Remove TOXENV from .travis.yml unless needed --- .travis.yml | 13 +++++-------- 1 file changed, 5 insertions(+), 8 deletions(-) diff --git a/.travis.yml b/.travis.yml index 66e1a9617..b029d8bda 100644 --- a/.travis.yml +++ b/.travis.yml @@ -12,17 +12,14 @@ matrix: - env: TOXENV=flake8 python: 3.8 - env: TOXENV=pypy3 - - env: TOXENV=py35 - python: 3.5 + - python: 3.5 - env: TOXENV=pinned python: 3.5 - env: TOXENV=py35-asyncio python: 3.5.2 - - env: TOXENV=py36 - python: 3.6 - - env: TOXENV=py37 - python: 3.7 - - env: TOXENV=py38 + - python: 3.6 + - python: 3.7 + - env: PYPI_RELEASE_JOB=true python: 3.8 - env: TOXENV=extra-deps python: 3.8 @@ -62,4 +59,4 @@ deploy: on: tags: true repo: scrapy/scrapy - condition: "$TOXENV == py37 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" + condition: "$PYPI_RELEASE_JOB == true && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$" From 83d7360bb709cf2c73680260c58b767006f42b12 Mon Sep 17 00:00:00 2001 From: Mikhail Korobov Date: Mon, 4 May 2020 02:00:11 +0500 Subject: [PATCH 306/313] Don't mention unsupported package versions in docs --- docs/topics/settings.rst | 16 +++------------- 1 file changed, 3 insertions(+), 13 deletions(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index 18f81838f..e3da1bd12 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -420,10 +420,9 @@ connections (for ``HTTP10DownloadHandler``). .. note:: HTTP/1.0 is rarely used nowadays so you can safely ignore this setting, - unless you use Twisted<11.1, or if you really want to use HTTP/1.0 - and override :setting:`DOWNLOAD_HANDLERS_BASE` for ``http(s)`` scheme - accordingly, i.e. to - ``'scrapy.core.downloader.handlers.http.HTTP10DownloadHandler'``. + unless you really want to use HTTP/1.0 and override + :setting:`DOWNLOAD_HANDLERS_BASE` for ``http(s)`` scheme accordingly, + i.e. to ``'scrapy.core.downloader.handlers.http.HTTP10DownloadHandler'``. .. setting:: DOWNLOADER_CLIENTCONTEXTFACTORY @@ -447,7 +446,6 @@ or even enable client-side authentication (and various other things). Scrapy also has another context factory class that you can set, ``'scrapy.core.downloader.contextfactory.BrowserLikeContextFactory'``, which uses the platform's certificates to validate remote endpoints. - **This is only available if you use Twisted>=14.0.** If you do use a custom ContextFactory, make sure its ``__init__`` method accepts a ``method`` parameter (this is the ``OpenSSL.SSL`` method mapping @@ -494,10 +492,6 @@ This setting must be one of these string values: - ``'TLSv1.2'``: forces TLS version 1.2 - ``'SSLv3'``: forces SSL version 3 (**not recommended**) -.. note:: - - We recommend that you use PyOpenSSL>=0.13 and Twisted>=0.13 - or above (Twisted>=14.0 if you can). .. setting:: DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING @@ -660,8 +654,6 @@ If you want to disable it set to 0. spider attribute and per-request using :reqmeta:`download_maxsize` Request.meta key. - This feature needs Twisted >= 11.1. - .. setting:: DOWNLOAD_WARNSIZE DOWNLOAD_WARNSIZE @@ -679,8 +671,6 @@ If you want to disable it set to 0. spider attribute and per-request using :reqmeta:`download_warnsize` Request.meta key. - This feature needs Twisted >= 11.1. - .. setting:: DOWNLOAD_FAIL_ON_DATALOSS DOWNLOAD_FAIL_ON_DATALOSS From fe6154e4faee375e7f47d61ceafabde7a3289bf3 Mon Sep 17 00:00:00 2001 From: Mikhail Korobov Date: Mon, 4 May 2020 18:18:38 +0500 Subject: [PATCH 307/313] clarify DOWNLOADER_HTTPCLIENTFACTORY docs --- docs/topics/settings.rst | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/topics/settings.rst b/docs/topics/settings.rst index e3da1bd12..f06d9db3c 100644 --- a/docs/topics/settings.rst +++ b/docs/topics/settings.rst @@ -421,7 +421,7 @@ connections (for ``HTTP10DownloadHandler``). HTTP/1.0 is rarely used nowadays so you can safely ignore this setting, unless you really want to use HTTP/1.0 and override - :setting:`DOWNLOAD_HANDLERS_BASE` for ``http(s)`` scheme accordingly, + :setting:`DOWNLOAD_HANDLERS` for ``http(s)`` scheme accordingly, i.e. to ``'scrapy.core.downloader.handlers.http.HTTP10DownloadHandler'``. .. setting:: DOWNLOADER_CLIENTCONTEXTFACTORY From 17c0cf64aee1641e1ad33c5b46a61435c5969f2f Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta <1731933+elacuesta@users.noreply.github.com> Date: Tue, 5 May 2020 19:14:48 -0300 Subject: [PATCH 308/313] Flake8: remove W504 code (#4525) Co-authored-by: Mikhail Korobov --- pytest.ini | 16 ++++++++-------- scrapy/contracts/__init__.py | 4 ++-- scrapy/downloadermiddlewares/redirect.py | 11 +++++++---- scrapy/extensions/telnet.py | 6 ++++-- scrapy/linkextractors/__init__.py | 3 +-- scrapy/spidermiddlewares/referer.py | 14 ++++++++------ scrapy/utils/gz.py | 3 +-- tests/test_utils_http.py | 8 ++++---- 8 files changed, 35 insertions(+), 30 deletions(-) diff --git a/pytest.ini b/pytest.ini index e8911ee3f..4f3494e0e 100644 --- a/pytest.ini +++ b/pytest.ini @@ -44,12 +44,12 @@ flake8-ignore = scrapy/commands/startproject.py E127 E501 E128 scrapy/commands/version.py E501 E128 # scrapy/contracts - scrapy/contracts/__init__.py E501 W504 + scrapy/contracts/__init__.py E501 scrapy/contracts/default.py E128 # scrapy/core scrapy/core/engine.py E501 E128 E127 scrapy/core/scheduler.py E501 - scrapy/core/scraper.py E501 E128 W504 + scrapy/core/scraper.py E501 E128 scrapy/core/spidermw.py E501 E126 scrapy/core/downloader/__init__.py E501 scrapy/core/downloader/contextfactory.py E501 E128 E126 @@ -68,7 +68,7 @@ flake8-ignore = scrapy/downloadermiddlewares/httpcache.py E501 E126 scrapy/downloadermiddlewares/httpcompression.py E501 E128 scrapy/downloadermiddlewares/httpproxy.py E501 - scrapy/downloadermiddlewares/redirect.py E501 W504 + scrapy/downloadermiddlewares/redirect.py E501 scrapy/downloadermiddlewares/retry.py E501 E126 scrapy/downloadermiddlewares/robotstxt.py E501 scrapy/downloadermiddlewares/stats.py E501 @@ -79,7 +79,7 @@ flake8-ignore = scrapy/extensions/httpcache.py E128 E501 scrapy/extensions/memdebug.py E501 scrapy/extensions/spiderstate.py E501 - scrapy/extensions/telnet.py E501 W504 + scrapy/extensions/telnet.py E501 scrapy/extensions/throttle.py E501 # scrapy/http scrapy/http/common.py E501 @@ -90,7 +90,7 @@ flake8-ignore = scrapy/http/response/__init__.py E501 E128 scrapy/http/response/text.py E501 E128 E124 # scrapy/linkextractors - scrapy/linkextractors/__init__.py E501 E402 W504 + scrapy/linkextractors/__init__.py E501 E402 scrapy/linkextractors/lxmlhtml.py E501 # scrapy/loader scrapy/loader/__init__.py E501 E128 @@ -110,7 +110,7 @@ flake8-ignore = # scrapy/spidermiddlewares scrapy/spidermiddlewares/httperror.py E501 scrapy/spidermiddlewares/offsite.py E501 - scrapy/spidermiddlewares/referer.py E501 E129 W504 + scrapy/spidermiddlewares/referer.py E501 E129 scrapy/spidermiddlewares/urllength.py E501 # scrapy/spiders scrapy/spiders/__init__.py E501 E402 @@ -125,7 +125,7 @@ flake8-ignore = scrapy/utils/decorators.py E501 scrapy/utils/defer.py E501 E128 scrapy/utils/deprecate.py E128 E501 E127 - scrapy/utils/gz.py E501 W504 + scrapy/utils/gz.py E501 scrapy/utils/http.py F403 scrapy/utils/httpobj.py E501 scrapy/utils/iterators.py E501 @@ -234,7 +234,7 @@ flake8-ignore = tests/test_utils_datatypes.py E402 E501 tests/test_utils_defer.py E501 F841 tests/test_utils_deprecate.py F841 E501 - tests/test_utils_http.py E501 E128 W504 + tests/test_utils_http.py E501 E128 tests/test_utils_iterators.py E501 E128 E129 tests/test_utils_log.py E741 tests/test_utils_python.py E501 diff --git a/scrapy/contracts/__init__.py b/scrapy/contracts/__init__.py index 41d4f25b2..5af3831a2 100644 --- a/scrapy/contracts/__init__.py +++ b/scrapy/contracts/__init__.py @@ -17,10 +17,10 @@ class ContractsManager: self.contracts[contract.name] = contract def tested_methods_from_spidercls(self, spidercls): + is_method = re.compile(r"^\s*@", re.MULTILINE).search methods = [] for key, value in getmembers(spidercls): - if (callable(value) and value.__doc__ and - re.search(r'^\s*@', value.__doc__, re.MULTILINE)): + if callable(value) and value.__doc__ and is_method(value.__doc__): methods.append(key) return methods diff --git a/scrapy/downloadermiddlewares/redirect.py b/scrapy/downloadermiddlewares/redirect.py index 09ee8377e..b32afb8e4 100644 --- a/scrapy/downloadermiddlewares/redirect.py +++ b/scrapy/downloadermiddlewares/redirect.py @@ -60,11 +60,14 @@ class RedirectMiddleware(BaseRedirectMiddleware): Handle redirection of requests based on response status and meta-refresh html tag. """ + def process_response(self, request, response, spider): - if (request.meta.get('dont_redirect', False) or - response.status in getattr(spider, 'handle_httpstatus_list', []) or - response.status in request.meta.get('handle_httpstatus_list', []) or - request.meta.get('handle_httpstatus_all', False)): + if ( + request.meta.get('dont_redirect', False) + or response.status in getattr(spider, 'handle_httpstatus_list', []) + or response.status in request.meta.get('handle_httpstatus_list', []) + or request.meta.get('handle_httpstatus_all', False) + ): return response allowed_status = (301, 302, 303, 307, 308) diff --git a/scrapy/extensions/telnet.py b/scrapy/extensions/telnet.py index 04ffd7235..1663604e7 100644 --- a/scrapy/extensions/telnet.py +++ b/scrapy/extensions/telnet.py @@ -76,8 +76,10 @@ class TelnetConsole(protocol.ServerFactory): """An implementation of IPortal""" @defers def login(self_, credentials, mind, *interfaces): - if not (credentials.username == self.username.encode('utf8') and - credentials.checkPassword(self.password.encode('utf8'))): + if not ( + credentials.username == self.username.encode('utf8') + and credentials.checkPassword(self.password.encode('utf8')) + ): raise ValueError("Invalid credentials") protocol = telnet.TelnetBootstrapProtocol( diff --git a/scrapy/linkextractors/__init__.py b/scrapy/linkextractors/__init__.py index d0b5066b6..ae019c70f 100644 --- a/scrapy/linkextractors/__init__.py +++ b/scrapy/linkextractors/__init__.py @@ -61,8 +61,7 @@ class FilteringLinkExtractor: def __new__(cls, *args, **kwargs): from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor - if (issubclass(cls, FilteringLinkExtractor) and - not issubclass(cls, LxmlLinkExtractor)): + if issubclass(cls, FilteringLinkExtractor) and not issubclass(cls, LxmlLinkExtractor): warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, ' 'please use scrapy.linkextractors.LinkExtractor instead', ScrapyDeprecationWarning, stacklevel=2) diff --git a/scrapy/spidermiddlewares/referer.py b/scrapy/spidermiddlewares/referer.py index 3784de885..434067b00 100644 --- a/scrapy/spidermiddlewares/referer.py +++ b/scrapy/spidermiddlewares/referer.py @@ -163,9 +163,10 @@ class StrictOriginPolicy(ReferrerPolicy): name = POLICY_STRICT_ORIGIN def referrer(self, response_url, request_url): - if ((self.tls_protected(response_url) and - self.potentially_trustworthy(request_url)) - or not self.tls_protected(response_url)): + if ( + self.tls_protected(response_url) and self.potentially_trustworthy(request_url) + or not self.tls_protected(response_url) + ): return self.origin_referrer(response_url) @@ -213,9 +214,10 @@ class StrictOriginWhenCrossOriginPolicy(ReferrerPolicy): origin = self.origin(response_url) if origin == self.origin(request_url): return self.stripped_referrer(response_url) - elif ((self.tls_protected(response_url) and - self.potentially_trustworthy(request_url)) - or not self.tls_protected(response_url)): + elif ( + self.tls_protected(response_url) and self.potentially_trustworthy(request_url) + or not self.tls_protected(response_url) + ): return self.origin_referrer(response_url) diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index c291ae237..fbd7bd18f 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -52,8 +52,7 @@ def is_gzipped(response): """Return True if the response is gzipped, or False otherwise""" ctype = response.headers.get('Content-Type', b'') cenc = response.headers.get('Content-Encoding', b'').lower() - return (_is_gzipped(ctype) or - (_is_octetstream(ctype) and cenc in (b'gzip', b'x-gzip'))) + return _is_gzipped(ctype) or _is_octetstream(ctype) and cenc in (b'gzip', b'x-gzip') def gzip_magic_number(response): diff --git a/tests/test_utils_http.py b/tests/test_utils_http.py index 2fac3da1f..363b015a8 100644 --- a/tests/test_utils_http.py +++ b/tests/test_utils_http.py @@ -13,7 +13,7 @@ class ChunkedTest(unittest.TestCase): chunked_body += "8\r\n" + "sequence\r\n" chunked_body += "0\r\n\r\n" body = decode_chunked_transfer(chunked_body) - self.assertEqual(body, - "This is the data in the first chunk\r\n" + - "and this is the second one\r\n" + - "consequence") + self.assertEqual( + body, + "This is the data in the first chunk\r\nand this is the second one\r\nconsequence" + ) From 49e8a337f78ec5e30eacfcd201b66d68deeecb56 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 6 May 2020 09:37:01 -0300 Subject: [PATCH 309/313] Flake8: remove E127 (continuation line over-indented for visual indent) --- pytest.ini | 10 +++++----- scrapy/core/downloader/handlers/ftp.py | 11 ++++++----- scrapy/core/engine.py | 3 +-- scrapy/utils/deprecate.py | 21 +++++++++++---------- scrapy/utils/request.py | 3 +-- 5 files changed, 24 insertions(+), 24 deletions(-) diff --git a/pytest.ini b/pytest.ini index 4f3494e0e..fa65a0da2 100644 --- a/pytest.ini +++ b/pytest.ini @@ -41,13 +41,13 @@ flake8-ignore = scrapy/commands/runspider.py E501 scrapy/commands/settings.py E128 scrapy/commands/shell.py E128 E501 - scrapy/commands/startproject.py E127 E501 E128 + scrapy/commands/startproject.py E501 E128 scrapy/commands/version.py E501 E128 # scrapy/contracts scrapy/contracts/__init__.py E501 scrapy/contracts/default.py E128 # scrapy/core - scrapy/core/engine.py E501 E128 E127 + scrapy/core/engine.py E501 E128 scrapy/core/scheduler.py E501 scrapy/core/scraper.py E501 E128 scrapy/core/spidermw.py E501 E126 @@ -57,7 +57,7 @@ flake8-ignore = scrapy/core/downloader/tls.py E501 scrapy/core/downloader/webclient.py E501 E128 E126 scrapy/core/downloader/handlers/__init__.py E501 - scrapy/core/downloader/handlers/ftp.py E501 E128 E127 + scrapy/core/downloader/handlers/ftp.py E501 E128 scrapy/core/downloader/handlers/http10.py E501 scrapy/core/downloader/handlers/http11.py E501 scrapy/core/downloader/handlers/s3.py E501 E128 E126 @@ -124,7 +124,7 @@ flake8-ignore = scrapy/utils/datatypes.py E501 scrapy/utils/decorators.py E501 scrapy/utils/defer.py E501 E128 - scrapy/utils/deprecate.py E128 E501 E127 + scrapy/utils/deprecate.py E501 scrapy/utils/gz.py E501 scrapy/utils/http.py F403 scrapy/utils/httpobj.py E501 @@ -137,7 +137,7 @@ flake8-ignore = scrapy/utils/python.py E501 scrapy/utils/reactor.py E501 scrapy/utils/reqser.py E501 - scrapy/utils/request.py E127 E501 + scrapy/utils/request.py E501 scrapy/utils/response.py E501 E128 scrapy/utils/signal.py E501 E128 scrapy/utils/sitemap.py E501 diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index 432cb1831..94b55c347 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -94,11 +94,12 @@ class FTPDownloadHandler: def gotClient(self, client, request, filepath): self.client = client protocol = ReceivedDataProtocol(request.meta.get("ftp_local_filename")) - return client.retrieveFile(filepath, protocol)\ - .addCallbacks(callback=self._build_response, - callbackArgs=(request, protocol), - errback=self._failed, - errbackArgs=(request,)) + return client.retrieveFile(filepath, protocol).addCallbacks( + callback=self._build_response, + callbackArgs=(request, protocol), + errback=self._failed, + errbackArgs=(request,), + ) def _build_response(self, result, request, protocol): self.result = result diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index 77d71846e..324d21716 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -230,8 +230,7 @@ class ExecutionEngine: def _downloaded(self, response, slot, request, spider): slot.remove_request(request) - return self.download(response, spider) \ - if isinstance(response, Request) else response + return self.download(response, spider) if isinstance(response, Request) else response def _download(self, request, spider): slot = self.slot diff --git a/scrapy/utils/deprecate.py b/scrapy/utils/deprecate.py index 36001d982..3dbea5fee 100644 --- a/scrapy/utils/deprecate.py +++ b/scrapy/utils/deprecate.py @@ -15,16 +15,17 @@ def attribute(obj, oldattr, newattr, version='0.12'): stacklevel=3) -def create_deprecated_class(name, new_class, clsdict=None, - warn_category=ScrapyDeprecationWarning, - warn_once=True, - old_class_path=None, - new_class_path=None, - subclass_warn_message="{cls} inherits from " - "deprecated class {old}, please inherit " - "from {new}.", - instance_warn_message="{cls} is deprecated, " - "instantiate {new} instead."): +def create_deprecated_class( + name, + new_class, + clsdict=None, + warn_category=ScrapyDeprecationWarning, + warn_once=True, + old_class_path=None, + new_class_path=None, + subclass_warn_message="{cls} inherits from deprecated class {old}, please inherit from {new}.", + instance_warn_message="{cls} is deprecated, instantiate {new} instead." +): """ Return a "deprecated" class that causes its subclasses to issue a warning. Subclasses of ``new_class`` are considered subclasses of this class. diff --git a/scrapy/utils/request.py b/scrapy/utils/request.py index b8c140a7e..12c03d78e 100644 --- a/scrapy/utils/request.py +++ b/scrapy/utils/request.py @@ -50,8 +50,7 @@ def request_fingerprint(request, include_headers=None, keep_fragments=False): """ if include_headers: - include_headers = tuple(to_bytes(h.lower()) - for h in sorted(include_headers)) + include_headers = tuple(to_bytes(h.lower()) for h in sorted(include_headers)) cache = _fingerprint_cache.setdefault(request, {}) cache_key = (include_headers, keep_fragments) if cache_key not in cache: From fe0c582ee083ad8085a33443af0ffbc67b44fc16 Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 6 May 2020 09:49:10 -0300 Subject: [PATCH 310/313] Flake8: remove E127 in tests (continuation line over-indented for visual indent) --- pytest.ini | 18 +-- tests/spiders.py | 3 +- tests/test_closespider.py | 3 +- tests/test_downloader_handlers.py | 9 +- ...test_downloadermiddleware_decompression.py | 4 +- tests/test_downloadermiddleware_redirect.py | 3 +- tests/test_http_request.py | 6 +- tests/test_selector.py | 3 +- tests/test_spidermiddleware_httperror.py | 6 +- tests/test_utils_url.py | 120 ++++++++++-------- 10 files changed, 90 insertions(+), 85 deletions(-) diff --git a/pytest.ini b/pytest.ini index fa65a0da2..3eefe70f1 100644 --- a/pytest.ini +++ b/pytest.ini @@ -171,8 +171,8 @@ flake8-ignore = tests/__init__.py E402 E501 tests/mockserver.py E401 E501 E126 E123 tests/pipelines.py F841 - tests/spiders.py E501 E127 - tests/test_closespider.py E501 E127 + tests/spiders.py E501 + tests/test_closespider.py E501 tests/test_command_fetch.py E501 tests/test_command_parse.py E501 E128 tests/test_command_shell.py E501 E128 @@ -181,17 +181,17 @@ flake8-ignore = tests/test_crawl.py E501 E741 tests/test_crawler.py F841 E501 tests/test_dependencies.py F841 E501 - tests/test_downloader_handlers.py E124 E127 E128 E501 E126 E123 + tests/test_downloader_handlers.py E124 E128 E501 E126 E123 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 tests/test_downloadermiddleware_cookies.py E741 E501 E128 E126 - tests/test_downloadermiddleware_decompression.py E127 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 tests/test_downloadermiddleware_httpcache.py E501 tests/test_downloadermiddleware_httpcompression.py E501 E126 E123 + tests/test_downloadermiddleware_decompression.py E501 tests/test_downloadermiddleware_httpproxy.py E501 E128 - tests/test_downloadermiddleware_redirect.py E501 E128 E127 + tests/test_downloadermiddleware_redirect.py E501 E128 tests/test_downloadermiddleware_retry.py E501 E128 E126 tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 @@ -202,7 +202,7 @@ flake8-ignore = tests/test_feedexport.py E501 F841 tests/test_http_cookies.py E501 tests/test_http_headers.py E501 - tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123 + tests/test_http_request.py E402 E501 E128 E128 E126 E123 tests/test_http_response.py E501 E128 tests/test_item.py E128 F841 tests/test_link.py E501 @@ -220,10 +220,10 @@ flake8-ignore = tests/test_responsetypes.py E501 tests/test_robotstxt_interface.py E501 E501 tests/test_scheduler.py E501 E126 E123 - tests/test_selector.py E501 E127 + tests/test_selector.py E501 tests/test_spider.py E501 tests/test_spidermiddleware.py E501 - tests/test_spidermiddleware_httperror.py E128 E501 E127 E121 + tests/test_spidermiddleware_httperror.py E128 E501 E121 tests/test_spidermiddleware_offsite.py E501 E128 E111 tests/test_spidermiddleware_output_chain.py E501 tests/test_spidermiddleware_referer.py E501 F841 E125 E124 E501 E121 @@ -243,7 +243,7 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_url.py E501 E127 E125 E501 E126 E123 + tests/test_utils_url.py E501 E125 E501 E126 E123 tests/test_webclient.py E501 E128 E122 E402 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 diff --git a/tests/spiders.py b/tests/spiders.py index 284c77829..33d5d02e1 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -184,8 +184,7 @@ class BrokenStartRequestsSpider(FollowAllSpider): if self.fail_yielding: 2 / 0 - assert self.seedsseen, \ - 'All start requests consumed before any download happened' + assert self.seedsseen, 'All start requests consumed before any download happened' def parse(self, response): self.seedsseen.append(response.meta.get('seed')) diff --git a/tests/test_closespider.py b/tests/test_closespider.py index 4a56425b7..5ec5e2989 100644 --- a/tests/test_closespider.py +++ b/tests/test_closespider.py @@ -41,8 +41,7 @@ class TestCloseSpider(TestCase): yield crawler.crawl(total=1000000, mockserver=self.mockserver) reason = crawler.spider.meta['close_reason'] self.assertEqual(reason, 'closespider_errorcount') - key = 'spider_exceptions/{name}'\ - .format(name=crawler.spider.exception_cls.__name__) + key = 'spider_exceptions/{name}'.format(name=crawler.spider.exception_cls.__name__) errorcount = crawler.stats.get_value(key) self.assertTrue(errorcount >= close_on) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 29d06bab4..24ef560c1 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -1090,8 +1090,7 @@ class DataURITestCase(unittest.TestCase): def test_default_mediatype_encoding(self): def _test(response): self.assertEqual(response.text, 'A brief note') - self.assertEqual(type(response), - responsetypes.from_mimetype("text/plain")) + self.assertEqual(type(response), responsetypes.from_mimetype("text/plain")) self.assertEqual(response.encoding, "US-ASCII") request = Request("data:,A%20brief%20note") @@ -1100,8 +1099,7 @@ class DataURITestCase(unittest.TestCase): def test_default_mediatype(self): def _test(response): self.assertEqual(response.text, u'\u038e\u03a3\u038e') - self.assertEqual(type(response), - responsetypes.from_mimetype("text/plain")) + self.assertEqual(type(response), responsetypes.from_mimetype("text/plain")) self.assertEqual(response.encoding, "iso-8859-7") request = Request("data:;charset=iso-8859-7,%be%d3%be") @@ -1119,8 +1117,7 @@ class DataURITestCase(unittest.TestCase): def test_mediatype_parameters(self): def _test(response): self.assertEqual(response.text, u'\u038e\u03a3\u038e') - self.assertEqual(type(response), - responsetypes.from_mimetype("text/plain")) + self.assertEqual(type(response), responsetypes.from_mimetype("text/plain")) self.assertEqual(response.encoding, "utf-8") request = Request('data:text/plain;foo=%22foo;bar%5C%22%22;' diff --git a/tests/test_downloadermiddleware_decompression.py b/tests/test_downloadermiddleware_decompression.py index 77b35a8c3..dbae4d3ae 100644 --- a/tests/test_downloadermiddleware_decompression.py +++ b/tests/test_downloadermiddleware_decompression.py @@ -28,8 +28,8 @@ class DecompressionMiddlewareTest(TestCase): for fmt in self.test_formats: rsp = self.test_responses[fmt] new = self.mw.process_response(None, rsp, self.spider) - assert isinstance(new, XmlResponse), \ - 'Failed %s, response type %s' % (fmt, type(new).__name__) + error_msg = 'Failed %s, response type %s' % (fmt, type(new).__name__) + assert isinstance(new, XmlResponse), error_msg assert_samelines(self, new.body, self.uncompressed_body, fmt) def test_plain_response(self): diff --git a/tests/test_downloadermiddleware_redirect.py b/tests/test_downloadermiddleware_redirect.py index 053e26fc3..551e124ab 100644 --- a/tests/test_downloadermiddleware_redirect.py +++ b/tests/test_downloadermiddleware_redirect.py @@ -181,8 +181,7 @@ class RedirectMiddlewareTest(unittest.TestCase): rsp = Response(url, headers={'Location': url2}, status=301, request=req) r = self.mw.process_response(req, rsp, self.spider) self.assertIs(r, rsp) - _test_passthrough(Request(url, meta={'handle_httpstatus_list': - [404, 301, 302]})) + _test_passthrough(Request(url, meta={'handle_httpstatus_list': [404, 301, 302]})) _test_passthrough(Request(url, meta={'handle_httpstatus_all': True})) def test_latin1_location(self): diff --git a/tests/test_http_request.py b/tests/test_http_request.py index cc2cddda4..b12841ba2 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -399,8 +399,7 @@ class FormRequestTest(RequestTest): def test_custom_encoding_bytes(self): data = {b'\xb5 one': b'two', b'price': b'\xa3 100'} - r2 = self.request_class("http://www.example.com", formdata=data, - encoding='latin1') + r2 = self.request_class("http://www.example.com", formdata=data, encoding='latin1') self.assertEqual(r2.method, 'POST') self.assertEqual(r2.encoding, 'latin1') self.assertQueryEqual(r2.body, b'price=%A3+100&%B5+one=two') @@ -408,8 +407,7 @@ class FormRequestTest(RequestTest): def test_custom_encoding_textual_data(self): data = {'price': u'£ 100'} - r3 = self.request_class("http://www.example.com", formdata=data, - encoding='latin1') + r3 = self.request_class("http://www.example.com", formdata=data, encoding='latin1') self.assertEqual(r3.encoding, 'latin1') self.assertEqual(r3.body, b'price=%A3+100') diff --git a/tests/test_selector.py b/tests/test_selector.py index 09c2546fb..65b0f5860 100644 --- a/tests/test_selector.py +++ b/tests/test_selector.py @@ -67,8 +67,7 @@ class SelectorTestCase(unittest.TestCase): headers = {'Content-Type': ['text/html; charset=utf-8']} response = HtmlResponse(url="http://example.com", headers=headers, body=html_utf8) x = Selector(response) - self.assertEqual(x.xpath("//span[@id='blank']/text()").getall(), - [u'\xa3']) + self.assertEqual(x.xpath("//span[@id='blank']/text()").getall(), [u'\xa3']) def test_badly_encoded_body(self): # \xe9 alone isn't valid utf8 sequence diff --git a/tests/test_spidermiddleware_httperror.py b/tests/test_spidermiddleware_httperror.py index dacd0147f..6b61df56f 100644 --- a/tests/test_spidermiddleware_httperror.py +++ b/tests/test_spidermiddleware_httperror.py @@ -111,8 +111,7 @@ class TestHttpErrorMiddlewareSettings(TestCase): self.mw.process_spider_input(self.res402, self.spider)) def test_meta_overrides_settings(self): - request = Request('http://scrapytest.org', - meta={'handle_httpstatus_list': [404]}) + request = Request('http://scrapytest.org', meta={'handle_httpstatus_list': [404]}) res404 = self.res404.copy() res404.request = request res402 = self.res402.copy() @@ -146,8 +145,7 @@ class TestHttpErrorMiddlewareHandleAll(TestCase): self.mw.process_spider_input(self.res404, self.spider)) def test_meta_overrides_settings(self): - request = Request('http://scrapytest.org', - meta={'handle_httpstatus_list': [404]}) + request = Request('http://scrapytest.org', meta={'handle_httpstatus_list': [404]}) res404 = self.res404.copy() res404.request = request res402 = self.res402.copy() diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 72a16e9b1..3bb6d40db 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -77,108 +77,124 @@ class UrlUtilsTest(unittest.TestCase): class AddHttpIfNoScheme(unittest.TestCase): def test_add_scheme(self): - self.assertEqual(add_http_if_no_scheme('www.example.com'), - 'http://www.example.com') + self.assertEqual(add_http_if_no_scheme('www.example.com'), 'http://www.example.com') def test_without_subdomain(self): - self.assertEqual(add_http_if_no_scheme('example.com'), - 'http://example.com') + self.assertEqual(add_http_if_no_scheme('example.com'), 'http://example.com') def test_path(self): - self.assertEqual(add_http_if_no_scheme('www.example.com/some/page.html'), - 'http://www.example.com/some/page.html') + self.assertEqual( + add_http_if_no_scheme('www.example.com/some/page.html'), + 'http://www.example.com/some/page.html') def test_port(self): - self.assertEqual(add_http_if_no_scheme('www.example.com:80'), - 'http://www.example.com:80') + self.assertEqual( + add_http_if_no_scheme('www.example.com:80'), + 'http://www.example.com:80') def test_fragment(self): - self.assertEqual(add_http_if_no_scheme('www.example.com/some/page#frag'), - 'http://www.example.com/some/page#frag') + self.assertEqual( + add_http_if_no_scheme('www.example.com/some/page#frag'), + 'http://www.example.com/some/page#frag') def test_query(self): - self.assertEqual(add_http_if_no_scheme('www.example.com/do?a=1&b=2&c=3'), - 'http://www.example.com/do?a=1&b=2&c=3') + self.assertEqual( + add_http_if_no_scheme('www.example.com/do?a=1&b=2&c=3'), + 'http://www.example.com/do?a=1&b=2&c=3') def test_username_password(self): - self.assertEqual(add_http_if_no_scheme('username:password@www.example.com'), - 'http://username:password@www.example.com') + self.assertEqual( + add_http_if_no_scheme('username:password@www.example.com'), + 'http://username:password@www.example.com') def test_complete_url(self): - self.assertEqual(add_http_if_no_scheme('username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag'), - 'http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag') + self.assertEqual( + add_http_if_no_scheme('username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag'), + 'http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag') def test_preserve_http(self): - self.assertEqual(add_http_if_no_scheme('http://www.example.com'), - 'http://www.example.com') + self.assertEqual(add_http_if_no_scheme('http://www.example.com'), 'http://www.example.com') def test_preserve_http_without_subdomain(self): - self.assertEqual(add_http_if_no_scheme('http://example.com'), - 'http://example.com') + self.assertEqual( + add_http_if_no_scheme('http://example.com'), + 'http://example.com') def test_preserve_http_path(self): - self.assertEqual(add_http_if_no_scheme('http://www.example.com/some/page.html'), - 'http://www.example.com/some/page.html') + self.assertEqual( + add_http_if_no_scheme('http://www.example.com/some/page.html'), + 'http://www.example.com/some/page.html') def test_preserve_http_port(self): - self.assertEqual(add_http_if_no_scheme('http://www.example.com:80'), - 'http://www.example.com:80') + self.assertEqual( + add_http_if_no_scheme('http://www.example.com:80'), + 'http://www.example.com:80') def test_preserve_http_fragment(self): - self.assertEqual(add_http_if_no_scheme('http://www.example.com/some/page#frag'), - 'http://www.example.com/some/page#frag') + self.assertEqual( + add_http_if_no_scheme('http://www.example.com/some/page#frag'), + 'http://www.example.com/some/page#frag') def test_preserve_http_query(self): - self.assertEqual(add_http_if_no_scheme('http://www.example.com/do?a=1&b=2&c=3'), - 'http://www.example.com/do?a=1&b=2&c=3') + self.assertEqual( + add_http_if_no_scheme('http://www.example.com/do?a=1&b=2&c=3'), + 'http://www.example.com/do?a=1&b=2&c=3') def test_preserve_http_username_password(self): - self.assertEqual(add_http_if_no_scheme('http://username:password@www.example.com'), - 'http://username:password@www.example.com') + self.assertEqual( + add_http_if_no_scheme('http://username:password@www.example.com'), + 'http://username:password@www.example.com') def test_preserve_http_complete_url(self): - self.assertEqual(add_http_if_no_scheme('http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag'), - 'http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag') + self.assertEqual( + add_http_if_no_scheme('http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag'), + 'http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag') def test_protocol_relative(self): - self.assertEqual(add_http_if_no_scheme('//www.example.com'), - 'http://www.example.com') + self.assertEqual( + add_http_if_no_scheme('//www.example.com'), 'http://www.example.com') def test_protocol_relative_without_subdomain(self): - self.assertEqual(add_http_if_no_scheme('//example.com'), - 'http://example.com') + self.assertEqual( + add_http_if_no_scheme('//example.com'), 'http://example.com') def test_protocol_relative_path(self): - self.assertEqual(add_http_if_no_scheme('//www.example.com/some/page.html'), - 'http://www.example.com/some/page.html') + self.assertEqual( + add_http_if_no_scheme('//www.example.com/some/page.html'), + 'http://www.example.com/some/page.html') def test_protocol_relative_port(self): - self.assertEqual(add_http_if_no_scheme('//www.example.com:80'), - 'http://www.example.com:80') + self.assertEqual( + add_http_if_no_scheme('//www.example.com:80'), + 'http://www.example.com:80') def test_protocol_relative_fragment(self): - self.assertEqual(add_http_if_no_scheme('//www.example.com/some/page#frag'), - 'http://www.example.com/some/page#frag') + self.assertEqual( + add_http_if_no_scheme('//www.example.com/some/page#frag'), + 'http://www.example.com/some/page#frag') def test_protocol_relative_query(self): - self.assertEqual(add_http_if_no_scheme('//www.example.com/do?a=1&b=2&c=3'), - 'http://www.example.com/do?a=1&b=2&c=3') + self.assertEqual( + add_http_if_no_scheme('//www.example.com/do?a=1&b=2&c=3'), + 'http://www.example.com/do?a=1&b=2&c=3') def test_protocol_relative_username_password(self): - self.assertEqual(add_http_if_no_scheme('//username:password@www.example.com'), - 'http://username:password@www.example.com') + self.assertEqual( + add_http_if_no_scheme('//username:password@www.example.com'), + 'http://username:password@www.example.com') def test_protocol_relative_complete_url(self): - self.assertEqual(add_http_if_no_scheme('//username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag'), - 'http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag') + self.assertEqual( + add_http_if_no_scheme('//username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag'), + 'http://username:password@www.example.com:80/some/page/do?a=1&b=2&c=3#frag') def test_preserve_https(self): - self.assertEqual(add_http_if_no_scheme('https://www.example.com'), - 'https://www.example.com') + self.assertEqual( + add_http_if_no_scheme('https://www.example.com'), + 'https://www.example.com') def test_preserve_ftp(self): - self.assertEqual(add_http_if_no_scheme('ftp://www.example.com'), - 'ftp://www.example.com') + self.assertEqual(add_http_if_no_scheme('ftp://www.example.com'), 'ftp://www.example.com') class GuessSchemeTest(unittest.TestCase): From 63600243e08cb7e783798bd6c59fb97595488e9e Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 6 May 2020 10:21:01 -0300 Subject: [PATCH 311/313] Flake8: remove E125 (Continuation line with same indent as next logical line) Also remove E401 from pytest.ini - no occurrences in the codebase --- pytest.ini | 12 ++++---- scrapy/pipelines/media.py | 10 ++++--- tests/test_spidermiddleware_referer.py | 40 +++++++++++++------------- tests/test_utils_url.py | 14 ++++----- 4 files changed, 39 insertions(+), 37 deletions(-) diff --git a/pytest.ini b/pytest.ini index 3eefe70f1..8ed1ad0cf 100644 --- a/pytest.ini +++ b/pytest.ini @@ -35,7 +35,7 @@ flake8-ignore = scrapy/commands/check.py E501 scrapy/commands/crawl.py E501 scrapy/commands/edit.py E501 - scrapy/commands/fetch.py E401 E501 E128 + scrapy/commands/fetch.py E501 E128 scrapy/commands/genspider.py E128 E501 scrapy/commands/parse.py E128 E501 scrapy/commands/runspider.py E501 @@ -99,7 +99,7 @@ flake8-ignore = scrapy/pipelines/__init__.py E501 scrapy/pipelines/files.py E116 E501 scrapy/pipelines/images.py E501 - scrapy/pipelines/media.py E125 E501 + scrapy/pipelines/media.py E501 # scrapy/selector scrapy/selector/__init__.py F403 scrapy/selector/unified.py E501 E111 @@ -169,7 +169,7 @@ flake8-ignore = scrapy/statscollectors.py E501 # tests tests/__init__.py E402 E501 - tests/mockserver.py E401 E501 E126 E123 + tests/mockserver.py E501 E126 E123 tests/pipelines.py F841 tests/spiders.py E501 tests/test_closespider.py E501 @@ -196,7 +196,7 @@ flake8-ignore = tests/test_downloadermiddleware_robotstxt.py E501 tests/test_downloadermiddleware_stats.py E501 tests/test_dupefilters.py E501 E741 E128 E124 - tests/test_engine.py E401 E501 E128 + tests/test_engine.py E501 E128 tests/test_exporters.py E501 E128 E124 tests/test_extension_telnet.py F841 tests/test_feedexport.py E501 F841 @@ -226,7 +226,7 @@ flake8-ignore = tests/test_spidermiddleware_httperror.py E128 E501 E121 tests/test_spidermiddleware_offsite.py E501 E128 E111 tests/test_spidermiddleware_output_chain.py E501 - tests/test_spidermiddleware_referer.py E501 F841 E125 E124 E501 E121 + tests/test_spidermiddleware_referer.py E501 F841 E124 E501 E121 tests/test_squeues.py E501 E741 tests/test_utils_asyncio.py E501 tests/test_utils_conf.py E501 E128 @@ -243,7 +243,7 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_url.py E501 E125 E501 E126 E123 + tests/test_utils_url.py E501 E501 E126 E123 tests/test_webclient.py E501 E128 E122 E402 E123 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 diff --git a/scrapy/pipelines/media.py b/scrapy/pipelines/media.py index 8a0636264..aa65f4f0e 100644 --- a/scrapy/pipelines/media.py +++ b/scrapy/pipelines/media.py @@ -43,8 +43,7 @@ class MediaPipeline: if allow_redirects: self.handle_httpstatus_list = SequenceExclude(range(300, 400)) - def _key_for_pipe(self, key, base_class_name=None, - settings=None): + def _key_for_pipe(self, key, base_class_name=None, settings=None): """ >>> MediaPipeline()._key_for_pipe("IMAGES") 'IMAGES' @@ -55,8 +54,11 @@ class MediaPipeline: """ class_name = self.__class__.__name__ formatted_key = "{}_{}".format(class_name.upper(), key) - if class_name == base_class_name or not base_class_name \ - or (settings and not settings.get(formatted_key)): + if ( + not base_class_name + or class_name == base_class_name + or settings and not settings.get(formatted_key) + ): return key return formatted_key diff --git a/tests/test_spidermiddleware_referer.py b/tests/test_spidermiddleware_referer.py index 742adc64f..41589177a 100644 --- a/tests/test_spidermiddleware_referer.py +++ b/tests/test_spidermiddleware_referer.py @@ -478,32 +478,32 @@ class TestSettingsPolicyByName(TestCase): def test_valid_name(self): for s, p in [ - (POLICY_SCRAPY_DEFAULT, DefaultReferrerPolicy), - (POLICY_NO_REFERRER, NoReferrerPolicy), - (POLICY_NO_REFERRER_WHEN_DOWNGRADE, NoReferrerWhenDowngradePolicy), - (POLICY_SAME_ORIGIN, SameOriginPolicy), - (POLICY_ORIGIN, OriginPolicy), - (POLICY_STRICT_ORIGIN, StrictOriginPolicy), - (POLICY_ORIGIN_WHEN_CROSS_ORIGIN, OriginWhenCrossOriginPolicy), - (POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN, StrictOriginWhenCrossOriginPolicy), - (POLICY_UNSAFE_URL, UnsafeUrlPolicy), - ]: + (POLICY_SCRAPY_DEFAULT, DefaultReferrerPolicy), + (POLICY_NO_REFERRER, NoReferrerPolicy), + (POLICY_NO_REFERRER_WHEN_DOWNGRADE, NoReferrerWhenDowngradePolicy), + (POLICY_SAME_ORIGIN, SameOriginPolicy), + (POLICY_ORIGIN, OriginPolicy), + (POLICY_STRICT_ORIGIN, StrictOriginPolicy), + (POLICY_ORIGIN_WHEN_CROSS_ORIGIN, OriginWhenCrossOriginPolicy), + (POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN, StrictOriginWhenCrossOriginPolicy), + (POLICY_UNSAFE_URL, UnsafeUrlPolicy), + ]: settings = Settings({'REFERRER_POLICY': s}) mw = RefererMiddleware(settings) self.assertEqual(mw.default_policy, p) def test_valid_name_casevariants(self): for s, p in [ - (POLICY_SCRAPY_DEFAULT, DefaultReferrerPolicy), - (POLICY_NO_REFERRER, NoReferrerPolicy), - (POLICY_NO_REFERRER_WHEN_DOWNGRADE, NoReferrerWhenDowngradePolicy), - (POLICY_SAME_ORIGIN, SameOriginPolicy), - (POLICY_ORIGIN, OriginPolicy), - (POLICY_STRICT_ORIGIN, StrictOriginPolicy), - (POLICY_ORIGIN_WHEN_CROSS_ORIGIN, OriginWhenCrossOriginPolicy), - (POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN, StrictOriginWhenCrossOriginPolicy), - (POLICY_UNSAFE_URL, UnsafeUrlPolicy), - ]: + (POLICY_SCRAPY_DEFAULT, DefaultReferrerPolicy), + (POLICY_NO_REFERRER, NoReferrerPolicy), + (POLICY_NO_REFERRER_WHEN_DOWNGRADE, NoReferrerWhenDowngradePolicy), + (POLICY_SAME_ORIGIN, SameOriginPolicy), + (POLICY_ORIGIN, OriginPolicy), + (POLICY_STRICT_ORIGIN, StrictOriginPolicy), + (POLICY_ORIGIN_WHEN_CROSS_ORIGIN, OriginWhenCrossOriginPolicy), + (POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN, StrictOriginWhenCrossOriginPolicy), + (POLICY_UNSAFE_URL, UnsafeUrlPolicy), + ]: settings = Settings({'REFERRER_POLICY': s.upper()}) mw = RefererMiddleware(settings) self.assertEqual(mw.default_policy, p) diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index 3bb6d40db..bed1a5634 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -288,7 +288,7 @@ class StripUrl(unittest.TestCase): ('http://www.example.com', True, 'http://www.example.com/'), - ]: + ]: self.assertEqual(strip_url(input_url, origin_only=origin), output_url) def test_credentials(self): @@ -301,7 +301,7 @@ class StripUrl(unittest.TestCase): ('ftp://username:password@www.example.com/index.html?somekey=somevalue#section', 'ftp://www.example.com/index.html?somekey=somevalue'), - ]: + ]: self.assertEqual(strip_url(i, strip_credentials=True), o) def test_credentials_encoded_delims(self): @@ -320,7 +320,7 @@ class StripUrl(unittest.TestCase): # password: "user@domain.com" ('ftp://me:user%40domain.com@www.example.com/index.html?somekey=somevalue#section', 'ftp://www.example.com/index.html?somekey=somevalue'), - ]: + ]: self.assertEqual(strip_url(i, strip_credentials=True), o) def test_default_ports_creds_off(self): @@ -348,7 +348,7 @@ class StripUrl(unittest.TestCase): ('ftp://username:password@www.example.com:221/file.txt', 'ftp://www.example.com:221/file.txt'), - ]: + ]: self.assertEqual(strip_url(i), o) def test_default_ports(self): @@ -376,7 +376,7 @@ class StripUrl(unittest.TestCase): ('ftp://username:password@www.example.com:221/file.txt', 'ftp://username:password@www.example.com:221/file.txt'), - ]: + ]: self.assertEqual(strip_url(i, strip_default_port=True, strip_credentials=False), o) def test_default_ports_keep(self): @@ -404,7 +404,7 @@ class StripUrl(unittest.TestCase): ('ftp://username:password@www.example.com:221/file.txt', 'ftp://username:password@www.example.com:221/file.txt'), - ]: + ]: self.assertEqual(strip_url(i, strip_default_port=False, strip_credentials=False), o) def test_origin_only(self): @@ -420,7 +420,7 @@ class StripUrl(unittest.TestCase): ('https://username:password@www.example.com:443/index.html', 'https://www.example.com/'), - ]: + ]: self.assertEqual(strip_url(i, origin_only=True), o) From 628c4a531914b6803ae0ec4991363aad52069ca1 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Micha=C5=82=20Panek?= Date: Wed, 6 May 2020 17:09:20 +0200 Subject: [PATCH 312/313] Add a warning/error in case of incorrect gcs permissions (#4508) --- scrapy/pipelines/files.py | 14 ++++++++++++++ 1 file changed, 14 insertions(+) diff --git a/scrapy/pipelines/files.py b/scrapy/pipelines/files.py index ae365db5b..a9066986b 100644 --- a/scrapy/pipelines/files.py +++ b/scrapy/pipelines/files.py @@ -230,6 +230,20 @@ class GCSFilesStore: bucket, prefix = uri[5:].split('/', 1) self.bucket = client.bucket(bucket) self.prefix = prefix + permissions = self.bucket.test_iam_permissions( + ['storage.objects.get', 'storage.objects.create'] + ) + if 'storage.objects.get' not in permissions: + logger.warning( + "No 'storage.objects.get' permission for GSC bucket %(bucket)s. " + "Checking if files are up to date will be impossible. Files will be downloaded every time.", + {'bucket': bucket} + ) + if 'storage.objects.create' not in permissions: + logger.error( + "No 'storage.objects.create' permission for GSC bucket %(bucket)s. Saving files will be impossible!", + {'bucket': bucket} + ) def stat_file(self, path, info): def _onsuccess(blob): From 8643e8d3557449393989b15b9b8f2ec813f3e6ad Mon Sep 17 00:00:00 2001 From: Eugenio Lacuesta Date: Wed, 6 May 2020 12:26:04 -0300 Subject: [PATCH 313/313] Flake8: remove E123 (Closing bracket does not match indentation of opening bracket's line) --- pytest.ini | 18 +++--- scrapy/extensions/closespider.py | 2 +- scrapy/http/request/form.py | 11 ++-- tests/test_downloader_handlers.py | 19 +++--- ...st_downloadermiddleware_httpcompression.py | 24 +++---- tests/test_http_request.py | 2 +- tests/test_scheduler.py | 30 +++++---- tests/test_utils_url.py | 64 +++++++++++-------- tests/test_webclient.py | 6 +- 9 files changed, 95 insertions(+), 81 deletions(-) diff --git a/pytest.ini b/pytest.ini index 8ed1ad0cf..1a73b41be 100644 --- a/pytest.ini +++ b/pytest.ini @@ -73,7 +73,7 @@ flake8-ignore = scrapy/downloadermiddlewares/robotstxt.py E501 scrapy/downloadermiddlewares/stats.py E501 # scrapy/extensions - scrapy/extensions/closespider.py E501 E128 E123 + scrapy/extensions/closespider.py E501 E128 scrapy/extensions/corestats.py E501 scrapy/extensions/feedexport.py E128 E501 scrapy/extensions/httpcache.py E128 E501 @@ -85,7 +85,7 @@ flake8-ignore = scrapy/http/common.py E501 scrapy/http/cookies.py E501 scrapy/http/request/__init__.py E501 - scrapy/http/request/form.py E501 E123 + scrapy/http/request/form.py E501 scrapy/http/request/json_request.py E501 scrapy/http/response/__init__.py E501 E128 scrapy/http/response/text.py E501 E128 E124 @@ -169,7 +169,7 @@ flake8-ignore = scrapy/statscollectors.py E501 # tests tests/__init__.py E402 E501 - tests/mockserver.py E501 E126 E123 + tests/mockserver.py E501 E126 tests/pipelines.py F841 tests/spiders.py E501 tests/test_closespider.py E501 @@ -181,14 +181,14 @@ flake8-ignore = tests/test_crawl.py E501 E741 tests/test_crawler.py F841 E501 tests/test_dependencies.py F841 E501 - tests/test_downloader_handlers.py E124 E128 E501 E126 E123 + tests/test_downloader_handlers.py E124 E128 E501 E126 tests/test_downloadermiddleware.py E501 tests/test_downloadermiddleware_ajaxcrawlable.py E501 tests/test_downloadermiddleware_cookies.py E741 E501 E128 E126 tests/test_downloadermiddleware_defaultheaders.py E501 tests/test_downloadermiddleware_downloadtimeout.py E501 tests/test_downloadermiddleware_httpcache.py E501 - tests/test_downloadermiddleware_httpcompression.py E501 E126 E123 + tests/test_downloadermiddleware_httpcompression.py E501 E126 tests/test_downloadermiddleware_decompression.py E501 tests/test_downloadermiddleware_httpproxy.py E501 E128 tests/test_downloadermiddleware_redirect.py E501 E128 @@ -202,7 +202,7 @@ flake8-ignore = tests/test_feedexport.py E501 F841 tests/test_http_cookies.py E501 tests/test_http_headers.py E501 - tests/test_http_request.py E402 E501 E128 E128 E126 E123 + tests/test_http_request.py E402 E501 E128 E128 E126 tests/test_http_response.py E501 E128 tests/test_item.py E128 F841 tests/test_link.py E501 @@ -219,7 +219,7 @@ flake8-ignore = tests/test_request_cb_kwargs.py E501 tests/test_responsetypes.py E501 tests/test_robotstxt_interface.py E501 E501 - tests/test_scheduler.py E501 E126 E123 + tests/test_scheduler.py E501 E126 tests/test_selector.py E501 tests/test_spider.py E501 tests/test_spidermiddleware.py E501 @@ -243,8 +243,8 @@ flake8-ignore = tests/test_utils_response.py E501 tests/test_utils_signal.py E741 F841 tests/test_utils_sitemap.py E128 E501 E124 - tests/test_utils_url.py E501 E501 E126 E123 - tests/test_webclient.py E501 E128 E122 E402 E123 E126 + tests/test_utils_url.py E501 E501 E126 + tests/test_webclient.py E501 E128 E122 E402 E126 tests/test_cmdline/__init__.py E501 tests/test_settings/__init__.py E501 E128 tests/test_spiderloader/__init__.py E128 E501 diff --git a/scrapy/extensions/closespider.py b/scrapy/extensions/closespider.py index e3f212bef..812844c0a 100644 --- a/scrapy/extensions/closespider.py +++ b/scrapy/extensions/closespider.py @@ -20,7 +20,7 @@ class CloseSpider: 'itemcount': crawler.settings.getint('CLOSESPIDER_ITEMCOUNT'), 'pagecount': crawler.settings.getint('CLOSESPIDER_PAGECOUNT'), 'errorcount': crawler.settings.getint('CLOSESPIDER_ERRORCOUNT'), - } + } if not any(self.close_on.values()): raise NotConfigured diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py index af02c8484..cd4e3373f 100644 --- a/scrapy/http/request/form.py +++ b/scrapy/http/request/form.py @@ -178,12 +178,11 @@ def _get_clickable(clickdata, form): if the latter is given. If not, it returns the first clickable element found """ - clickables = [ - el for el in form.xpath( - 'descendant::input[re:test(@type, "^(submit|image)$", "i")]' - '|descendant::button[not(@type) or re:test(@type, "^submit$", "i")]', - namespaces={"re": "http://exslt.org/regular-expressions"}) - ] + clickables = list(form.xpath( + 'descendant::input[re:test(@type, "^(submit|image)$", "i")]' + '|descendant::button[not(@type) or re:test(@type, "^submit$", "i")]', + namespaces={"re": "http://exslt.org/regular-expressions"} + )) if not clickables: return diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 24ef560c1..f93bce8ef 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -822,11 +822,15 @@ class S3TestCase(unittest.TestCase): def test_request_signing2(self): # puts an object into the johnsmith bucket. date = 'Tue, 27 Mar 2007 21:15:45 +0000' - req = Request('s3://johnsmith/photos/puppy.jpg', method='PUT', headers={ - 'Content-Type': 'image/jpeg', - 'Date': date, - 'Content-Length': '94328', - }) + req = Request( + 's3://johnsmith/photos/puppy.jpg', + method='PUT', + headers={ + 'Content-Type': 'image/jpeg', + 'Date': date, + 'Content-Length': '94328', + }, + ) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) self.assertEqual(httpreq.headers['Authorization'], @@ -906,11 +910,10 @@ class S3TestCase(unittest.TestCase): # ensure that spaces are quoted properly before signing date = 'Tue, 27 Mar 2007 19:42:41 +0000' req = Request( - ("s3://johnsmith/photos/my puppy.jpg" - "?response-content-disposition=my puppy.jpg"), + "s3://johnsmith/photos/my puppy.jpg?response-content-disposition=my puppy.jpg", method='GET', headers={'Date': date}, - ) + ) with self._mocked_date(date): httpreq = self.download_request(req, self.spider) self.assertEqual( diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index 106ca3360..e86568bfb 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -16,12 +16,12 @@ from w3lib.encoding import resolve_encoding SAMPLEDIR = join(tests_datadir, 'compressed') FORMAT = { - 'gzip': ('html-gzip.bin', 'gzip'), - 'x-gzip': ('html-gzip.bin', 'gzip'), - 'rawdeflate': ('html-rawdeflate.bin', 'deflate'), - 'zlibdeflate': ('html-zlibdeflate.bin', 'deflate'), - 'br': ('html-br.bin', 'br') - } + 'gzip': ('html-gzip.bin', 'gzip'), + 'x-gzip': ('html-gzip.bin', 'gzip'), + 'rawdeflate': ('html-rawdeflate.bin', 'deflate'), + 'zlibdeflate': ('html-zlibdeflate.bin', 'deflate'), + 'br': ('html-br.bin', 'br'), +} class HttpCompressionTest(TestCase): @@ -40,12 +40,12 @@ class HttpCompressionTest(TestCase): body = sample.read() headers = { - 'Server': 'Yaws/1.49 Yet Another Web Server', - 'Date': 'Sun, 08 Mar 2009 00:41:03 GMT', - 'Content-Length': len(body), - 'Content-Type': 'text/html', - 'Content-Encoding': contentencoding, - } + 'Server': 'Yaws/1.49 Yet Another Web Server', + 'Date': 'Sun, 08 Mar 2009 00:41:03 GMT', + 'Content-Length': len(body), + 'Content-Type': 'text/html', + 'Content-Encoding': contentencoding, + } response = Response('http://scrapytest.org/', body=body, headers=headers) response.request = Request('http://scrapytest.org', headers={'Accept-Encoding': 'gzip, deflate'}) diff --git a/tests/test_http_request.py b/tests/test_http_request.py index b12841ba2..3b6d119a9 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -467,7 +467,7 @@ class FormRequestTest(RequestTest): """, url="http://www.example.com/this/list.html", encoding='latin1', - ) + ) req = self.request_class.from_response(response, formdata={'one': ['two', 'three'], 'six': 'seven'}) diff --git a/tests/test_scheduler.py b/tests/test_scheduler.py index 00568aee9..930a5dd99 100644 --- a/tests/test_scheduler.py +++ b/tests/test_scheduler.py @@ -46,13 +46,13 @@ class MockCrawler(Crawler): def __init__(self, priority_queue_cls, jobdir): settings = dict( - SCHEDULER_DEBUG=False, - SCHEDULER_DISK_QUEUE='scrapy.squeues.PickleLifoDiskQueue', - SCHEDULER_MEMORY_QUEUE='scrapy.squeues.LifoMemoryQueue', - SCHEDULER_PRIORITY_QUEUE=priority_queue_cls, - JOBDIR=jobdir, - DUPEFILTER_CLASS='scrapy.dupefilters.BaseDupeFilter' - ) + SCHEDULER_DEBUG=False, + SCHEDULER_DISK_QUEUE='scrapy.squeues.PickleLifoDiskQueue', + SCHEDULER_MEMORY_QUEUE='scrapy.squeues.LifoMemoryQueue', + SCHEDULER_PRIORITY_QUEUE=priority_queue_cls, + JOBDIR=jobdir, + DUPEFILTER_CLASS='scrapy.dupefilters.BaseDupeFilter', + ) super(MockCrawler, self).__init__(Spider, settings) self.engine = MockEngine(downloader=MockDownloader()) @@ -305,10 +305,12 @@ class StartUrlsSpider(Spider): class TestIntegrationWithDownloaderAwareInMemory(TestCase): def setUp(self): self.crawler = get_crawler( - StartUrlsSpider, - {'SCHEDULER_PRIORITY_QUEUE': 'scrapy.pqueues.DownloaderAwarePriorityQueue', - 'DUPEFILTER_CLASS': 'scrapy.dupefilters.BaseDupeFilter'} - ) + spidercls=StartUrlsSpider, + settings_dict={ + 'SCHEDULER_PRIORITY_QUEUE': 'scrapy.pqueues.DownloaderAwarePriorityQueue', + 'DUPEFILTER_CLASS': 'scrapy.dupefilters.BaseDupeFilter', + }, + ) @defer.inlineCallbacks def tearDown(self): @@ -329,9 +331,9 @@ class TestIncompatibility(unittest.TestCase): def _incompatible(self): settings = dict( - SCHEDULER_PRIORITY_QUEUE='scrapy.pqueues.DownloaderAwarePriorityQueue', - CONCURRENT_REQUESTS_PER_IP=1 - ) + SCHEDULER_PRIORITY_QUEUE='scrapy.pqueues.DownloaderAwarePriorityQueue', + CONCURRENT_REQUESTS_PER_IP=1, + ) crawler = Crawler(Spider, settings) scheduler = Scheduler.from_crawler(crawler) spider = Spider(name='spider') diff --git a/tests/test_utils_url.py b/tests/test_utils_url.py index bed1a5634..1f8388957 100644 --- a/tests/test_utils_url.py +++ b/tests/test_utils_url.py @@ -218,41 +218,49 @@ def create_skipped_scheme_t(args): return do_expected -for k, args in enumerate([ - ('/index', 'file://'), - ('/index.html', 'file://'), - ('./index.html', 'file://'), - ('../index.html', 'file://'), - ('../../index.html', 'file://'), - ('./data/index.html', 'file://'), - ('.hidden/data/index.html', 'file://'), - ('/home/user/www/index.html', 'file://'), - ('//home/user/www/index.html', 'file://'), - ('file:///home/user/www/index.html', 'file://'), +for k, args in enumerate( + [ + ('/index', 'file://'), + ('/index.html', 'file://'), + ('./index.html', 'file://'), + ('../index.html', 'file://'), + ('../../index.html', 'file://'), + ('./data/index.html', 'file://'), + ('.hidden/data/index.html', 'file://'), + ('/home/user/www/index.html', 'file://'), + ('//home/user/www/index.html', 'file://'), + ('file:///home/user/www/index.html', 'file://'), - ('index.html', 'http://'), - ('example.com', 'http://'), - ('www.example.com', 'http://'), - ('www.example.com/index.html', 'http://'), - ('http://example.com', 'http://'), - ('http://example.com/index.html', 'http://'), - ('localhost', 'http://'), - ('localhost/index.html', 'http://'), + ('index.html', 'http://'), + ('example.com', 'http://'), + ('www.example.com', 'http://'), + ('www.example.com/index.html', 'http://'), + ('http://example.com', 'http://'), + ('http://example.com/index.html', 'http://'), + ('localhost', 'http://'), + ('localhost/index.html', 'http://'), - # some corner cases (default to http://) - ('/', 'http://'), - ('.../test', 'http://'), - - ], start=1): + # some corner cases (default to http://) + ('/', 'http://'), + ('.../test', 'http://'), + ], + start=1, +): t_method = create_guess_scheme_t(args) t_method.__name__ = 'test_uri_%03d' % k setattr(GuessSchemeTest, t_method.__name__, t_method) # TODO: the following tests do not pass with current implementation -for k, args in enumerate([ - (r'C:\absolute\path\to\a\file.html', 'file://', - 'Windows filepath are not supported for scrapy shell'), - ], start=1): +for k, args in enumerate( + [ + ( + r'C:\absolute\path\to\a\file.html', + 'file://', + 'Windows filepath are not supported for scrapy shell', + ), + ], + start=1, +): t_method = create_skipped_scheme_t(args) t_method.__name__ = 'test_uri_skipped_%03d' % k setattr(GuessSchemeTest, t_method.__name__, t_method) diff --git a/tests/test_webclient.py b/tests/test_webclient.py index d4abebbfb..de61e2125 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -149,7 +149,8 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): headers={ 'X-Meta-Single': 'single', 'X-Meta-Multivalued': ['value1', 'value2'], - })) + }, + )) self._test(factory, b"GET /bar HTTP/1.0\r\n" @@ -165,7 +166,8 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): headers=Headers({ 'X-Meta-Single': 'single', 'X-Meta-Multivalued': ['value1', 'value2'], - }))) + }), + )) self._test(factory, b"GET /bar HTTP/1.0\r\n"