Merge branch 'master' into azure-pipelines

This commit is contained in:
Adrián Chaves 2020-07-14 11:26:19 +02:00 committed by GitHub
commit 6e119bd3e3
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
39 changed files with 378 additions and 166 deletions

View File

@ -468,7 +468,7 @@ Supported options:
* ``--callback`` or ``-c``: spider method to use as callback for parsing the
response
* ``--meta`` or ``-m``: additional request meta that will be passed to the callback
* ``--meta`` or ``-m``: additional request meta that will be passed to the callback
request. This must be a valid json string. Example: --meta='{"foo" : "bar"}'
* ``--cbkwargs``: additional keyword arguments that will be passed to the callback.
@ -491,6 +491,8 @@ Supported options:
* ``--verbose`` or ``-v``: display information for each depth level
* ``--output`` or ``-o``: dump scraped items to a file
.. skip: start
Usage example::

View File

@ -19,10 +19,12 @@ def _iter_command_classes(module_name):
# scrapy.utils.spider.iter_spider_classes
for module in walk_modules(module_name):
for obj in vars(module).values():
if inspect.isclass(obj) and \
issubclass(obj, ScrapyCommand) and \
obj.__module__ == module.__name__ and \
not obj == ScrapyCommand:
if (
inspect.isclass(obj)
and issubclass(obj, ScrapyCommand)
and obj.__module__ == module.__name__
and not obj == ScrapyCommand
):
yield obj

View File

@ -108,7 +108,7 @@ class ScrapyCommand:
class BaseRunSpiderCommand(ScrapyCommand):
"""
Common class used to share functionality between the crawl and runspider commands
Common class used to share functionality between the crawl, parse and runspider commands
"""
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)

View File

@ -26,6 +26,8 @@ class Command(BaseRunSpiderCommand):
else:
self.crawler_process.start()
if self.crawler_process.bootstrap_failed or \
(hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception):
if (
self.crawler_process.bootstrap_failed
or hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception
):
self.exitcode = 1

View File

@ -19,8 +19,10 @@ class Command(ScrapyCommand):
return "Fetch a URL using the Scrapy downloader"
def long_desc(self):
return "Fetch a URL using the Scrapy downloader and print its content " \
"to stdout. You may want to use --nolog to disable logging"
return (
"Fetch a URL using the Scrapy downloader and print its content"
" to stdout. You may want to use --nolog to disable logging"
)
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)

View File

@ -121,6 +121,7 @@ class Command(ScrapyCommand):
@property
def templates_dir(self):
_templates_base_dir = self.settings['TEMPLATES_DIR'] or \
join(scrapy.__path__[0], 'templates')
return join(_templates_base_dir, 'spiders')
return join(
self.settings['TEMPLATES_DIR'] or join(scrapy.__path__[0], 'templates'),
'spiders'
)

View File

@ -4,18 +4,16 @@ import logging
from itemadapter import is_item, ItemAdapter
from w3lib.url import is_url
from scrapy.commands import ScrapyCommand
from scrapy.commands import BaseRunSpiderCommand
from scrapy.http import Request
from scrapy.utils import display
from scrapy.utils.conf import arglist_to_dict
from scrapy.utils.spider import iterate_spider_output, spidercls_for_request
from scrapy.exceptions import UsageError
logger = logging.getLogger(__name__)
class Command(ScrapyCommand):
class Command(BaseRunSpiderCommand):
requires_project = True
spider = None
@ -31,11 +29,9 @@ class Command(ScrapyCommand):
return "Parse URL (using its spider) and print the results"
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)
BaseRunSpiderCommand.add_options(self, parser)
parser.add_option("--spider", dest="spider", default=None,
help="use this spider without looking for one")
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("--pipelines", action="store_true",
help="process items through pipelines")
parser.add_option("--nolinks", dest="nolinks", action="store_true",
@ -200,12 +196,15 @@ class Command(ScrapyCommand):
self.add_items(depth, items)
self.add_requests(depth, requests)
scraped_data = items if opts.output else []
if depth < opts.depth:
for req in requests:
req.meta['_depth'] = depth + 1
req.meta['_callback'] = req.callback
req.callback = callback
return requests
scraped_data += requests
return scraped_data
# update request meta if any extra meta was passed through the --meta/-m opts.
if opts.meta:
@ -221,18 +220,11 @@ class Command(ScrapyCommand):
return request
def process_options(self, args, opts):
ScrapyCommand.process_options(self, args, opts)
BaseRunSpiderCommand.process_options(self, args, opts)
self.process_spider_arguments(opts)
self.process_request_meta(opts)
self.process_request_cb_kwargs(opts)
def process_spider_arguments(self, opts):
try:
opts.spargs = arglist_to_dict(opts.spargs)
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
def process_request_meta(self, opts):
if opts.meta:
try:

View File

@ -1,10 +1,10 @@
import re
import os
import stat
import string
from importlib import import_module
from os.path import join, exists, abspath
from shutil import ignore_patterns, move, copy2, copystat
from stat import S_IWUSR as OWNER_WRITE_PERMISSION
import scrapy
from scrapy.commands import ScrapyCommand
@ -20,7 +20,12 @@ TEMPLATES_TO_RENDER = (
('${project_name}', 'middlewares.py.tmpl'),
)
IGNORE = ignore_patterns('*.pyc', '.svn')
IGNORE = ignore_patterns('*.pyc', '__pycache__', '.svn')
def _make_writable(path):
current_permissions = os.stat(path).st_mode
os.chmod(path, current_permissions | OWNER_WRITE_PERMISSION)
class Command(ScrapyCommand):
@ -78,30 +83,10 @@ class Command(ScrapyCommand):
self._copytree(srcname, dstname)
else:
copy2(srcname, dstname)
_make_writable(dstname)
copystat(src, dst)
self._set_rw_permissions(dst)
def _set_rw_permissions(self, path):
"""
Sets permissions of a directory tree to +rw and +rwx for folders.
This is necessary if the start template files come without write
permissions.
"""
mode_rw = (stat.S_IRUSR
| stat.S_IWUSR
| stat.S_IRGRP
| stat.S_IROTH)
mode_x = (stat.S_IXUSR
| stat.S_IXGRP
| stat.S_IXOTH)
os.chmod(path, mode_rw | mode_x)
for root, dirs, files in os.walk(path):
for dir in dirs:
os.chmod(join(root, dir), mode_rw | mode_x)
for file in files:
os.chmod(join(root, file), mode_rw)
_make_writable(dst)
def run(self, args, opts):
if len(args) not in (1, 2):
@ -137,6 +122,7 @@ class Command(ScrapyCommand):
@property
def templates_dir(self):
_templates_base_dir = self.settings['TEMPLATES_DIR'] or \
join(scrapy.__path__[0], 'templates')
return join(_templates_base_dir, 'project')
return join(
self.settings['TEMPLATES_DIR'] or join(scrapy.__path__[0], 'templates'),
'project'
)

View File

@ -8,8 +8,7 @@ class Command(fetch.Command):
return "Open URL in browser, as seen by Scrapy"
def long_desc(self):
return "Fetch a URL using the Scrapy downloader and show its " \
"contents in a browser"
return "Fetch a URL using the Scrapy downloader and show its contents in a browser"
def add_options(self, parser):
super(Command, self).add_options(parser)

View File

@ -141,10 +141,12 @@ class ExecutionEngine:
def _needs_backout(self, spider):
slot = self.slot
return not self.running \
or slot.closing \
or self.downloader.needs_backout() \
return (
not self.running
or slot.closing
or self.downloader.needs_backout()
or self.scraper.slot.needs_backout()
)
def _next_request_from_scheduler(self, spider):
slot = self.slot

View File

@ -33,10 +33,8 @@ class BaseRedirectMiddleware:
if ttl and redirects <= self.max_redirect_times:
redirected.meta['redirect_times'] = redirects
redirected.meta['redirect_ttl'] = ttl - 1
redirected.meta['redirect_urls'] = request.meta.get('redirect_urls', []) + \
[request.url]
redirected.meta['redirect_reasons'] = request.meta.get('redirect_reasons', []) + \
[reason]
redirected.meta['redirect_urls'] = request.meta.get('redirect_urls', []) + [request.url]
redirected.meta['redirect_reasons'] = request.meta.get('redirect_reasons', []) + [reason]
redirected.dont_filter = request.dont_filter
redirected.priority = request.priority + self.priority_adjust
logger.debug("Redirecting (%(reason)s) to %(redirected)s from %(request)s",
@ -99,8 +97,11 @@ class MetaRefreshMiddleware(BaseRedirectMiddleware):
self._maxdelay = settings.getint('METAREFRESH_MAXDELAY')
def process_response(self, request, response, spider):
if request.meta.get('dont_redirect', False) or request.method == 'HEAD' or \
not isinstance(response, HtmlResponse):
if (
request.meta.get('dont_redirect', False)
or request.method == 'HEAD'
or not isinstance(response, HtmlResponse)
):
return response
interval, url = get_meta_refresh(response,

View File

@ -60,8 +60,10 @@ class RetryMiddleware:
return response
def process_exception(self, request, exception, spider):
if isinstance(exception, self.EXCEPTIONS_TO_RETRY) \
and not request.meta.get('dont_retry', False):
if (
isinstance(exception, self.EXCEPTIONS_TO_RETRY)
and not request.meta.get('dont_retry', False)
):
return self._retry(request, exception, spider)
def _retry(self, request, reason, spider):

View File

@ -81,8 +81,10 @@ class MemoryUsage:
logger.error("Memory usage exceeded %(memusage)dM. Shutting down Scrapy...",
{'memusage': mem}, extra={'crawler': self.crawler})
if self.notify_mails:
subj = "%s terminated: memory usage exceeded %dM at %s" % \
(self.crawler.settings['BOT_NAME'], mem, socket.gethostname())
subj = (
"%s terminated: memory usage exceeded %dM at %s"
% (self.crawler.settings['BOT_NAME'], mem, socket.gethostname())
)
self._send_report(self.notify_mails, subj)
self.crawler.stats.set_value('memusage/limit_notified', 1)
@ -102,8 +104,10 @@ class MemoryUsage:
logger.warning("Memory usage reached %(memusage)dM",
{'memusage': mem}, extra={'crawler': self.crawler})
if self.notify_mails:
subj = "%s warning: memory usage reached %dM at %s" % \
(self.crawler.settings['BOT_NAME'], mem, socket.gethostname())
subj = (
"%s warning: memory usage reached %dM at %s"
% (self.crawler.settings['BOT_NAME'], mem, socket.gethostname())
)
self._send_report(self.notify_mails, subj)
self.crawler.stats.set_value('memusage/warning_notified', 1)
self.warned = True

View File

@ -205,8 +205,7 @@ def _get_clickable(clickdata, form):
# We didn't find it, so now we build an XPath expression out of the other
# arguments, because they can be used as such
xpath = u'.//*' + \
u''.join(u'[@%s="%s"]' % c for c in clickdata.items())
xpath = u'.//*' + u''.join(u'[@%s="%s"]' % c for c in clickdata.items())
el = form.xpath(xpath)
if len(el) == 1:
return (el[0].get('name'), el[0].get('value') or '')

View File

@ -62,8 +62,11 @@ class TextResponse(Response):
return self._declared_encoding() or self._body_inferred_encoding()
def _declared_encoding(self):
return self._encoding or self._headers_encoding() \
return (
self._encoding
or self._headers_encoding()
or self._body_declared_encoding()
)
def body_as_unicode(self):
"""Return body as unicode"""

View File

@ -21,12 +21,18 @@ class Link:
self.nofollow = nofollow
def __eq__(self, other):
return self.url == other.url and self.text == other.text and \
self.fragment == other.fragment and self.nofollow == other.nofollow
return (
self.url == other.url
and self.text == other.text
and self.fragment == other.fragment
and self.nofollow == other.nofollow
)
def __hash__(self):
return hash(self.url) ^ hash(self.text) ^ hash(self.fragment) ^ hash(self.nofollow)
def __repr__(self):
return 'Link(url=%r, text=%r, fragment=%r, nofollow=%r)' % \
(self.url, self.text, self.fragment, self.nofollow)
return (
'Link(url=%r, text=%r, fragment=%r, nofollow=%r)'
% (self.url, self.text, self.fragment, self.nofollow)
)

View File

@ -52,8 +52,7 @@ class SettingsAttribute:
self.priority = priority
def __str__(self):
return "<SettingsAttribute value={self.value!r} " \
"priority={self.priority}>".format(self=self)
return "<SettingsAttribute value={self.value!r} priority={self.priority}>".format(self=self)
__repr__ = __str__

View File

@ -28,8 +28,7 @@ class Root(Resource):
def _getarg(request, name, default=None, type=str):
return type(request.args[name][0]) \
if name in request.args else default
return type(request.args[name][0]) if name in request.args else default
if __name__ == '__main__':

View File

@ -101,11 +101,13 @@ def get_config(use_closest=True):
def get_sources(use_closest=True):
xdg_config_home = os.environ.get('XDG_CONFIG_HOME') or \
os.path.expanduser('~/.config')
sources = ['/etc/scrapy.cfg', r'c:\scrapy\scrapy.cfg',
xdg_config_home + '/scrapy.cfg',
os.path.expanduser('~/.scrapy.cfg')]
xdg_config_home = os.environ.get('XDG_CONFIG_HOME') or os.path.expanduser('~/.config')
sources = [
'/etc/scrapy.cfg',
r'c:\scrapy\scrapy.cfg',
xdg_config_home + '/scrapy.cfg',
os.path.expanduser('~/.scrapy.cfg'),
]
if use_closest:
sources.append(closest_scrapy_cfg())
return sources

View File

@ -9,8 +9,7 @@ from w3lib.http import basic_auth_header
class CurlParser(argparse.ArgumentParser):
def error(self, message):
error_msg = \
'There was an error parsing the curl command: {}'.format(message)
error_msg = 'There was an error parsing the curl command: {}'.format(message)
raise ValueError(error_msg)

View File

@ -18,8 +18,7 @@ def install_shutdown_handlers(function, override_sigint=True):
from twisted.internet import reactor
reactor._handleSignals()
signal.signal(signal.SIGTERM, function)
if signal.getsignal(signal.SIGINT) == signal.default_int_handler or \
override_sigint:
if signal.getsignal(signal.SIGINT) == signal.default_int_handler or override_sigint:
signal.signal(signal.SIGINT, function)
# Catch Ctrl-Break in windows
if hasattr(signal, 'SIGBREAK'):

View File

@ -47,13 +47,17 @@ def response_httprepr(response):
is provided only for reference, since it's not the exact stream of bytes
that was received (that's not exposed by Twisted).
"""
s = b"HTTP/1.1 " + to_bytes(str(response.status)) + b" " + \
to_bytes(http.RESPONSES.get(response.status, b'')) + b"\r\n"
values = [
b"HTTP/1.1 ",
to_bytes(str(response.status)),
b" ",
to_bytes(http.RESPONSES.get(response.status, b'')),
b"\r\n",
]
if response.headers:
s += response.headers.to_string() + b"\r\n"
s += b"\r\n"
s += response.body
return s
values.extend([response.headers.to_string(), b"\r\n"])
values.extend([b"\r\n", response.body])
return b"".join(values)
def open_in_browser(response, _openfunc=webbrowser.open):

View File

@ -34,10 +34,12 @@ def iter_spider_classes(module):
from scrapy.spiders import Spider
for obj in vars(module).values():
if inspect.isclass(obj) and \
issubclass(obj, Spider) and \
obj.__module__ == module.__name__ and \
getattr(obj, 'name', None):
if (
inspect.isclass(obj)
and issubclass(obj, Spider)
and obj.__module__ == module.__name__
and getattr(obj, 'name', None)
):
yield obj

View File

@ -1,5 +1,5 @@
import os
from os.path import join, abspath
from os.path import join, abspath, isfile, exists
from twisted.internet import defer
from scrapy.utils.testsite import SiteTest
from scrapy.utils.testproc import ProcessTest
@ -218,3 +218,24 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
)
self.assertRegex(_textmode(out), r"""# Scraped Items -+\n\[\]""")
self.assertIn("""Cannot find a rule that matches""", _textmode(stderr))
@defer.inlineCallbacks
def test_output_flag(self):
"""Checks if a file was created successfully having
correct format containing correct data in it.
"""
file_name = 'data.json'
file_path = join(self.proj_path, file_name)
yield self.execute([
'--spider', self.spider_name,
'-c', 'parse',
'-o', file_name,
self.url('/html')
])
self.assertTrue(exists(file_path))
self.assertTrue(isfile(file_path))
content = '[\n{},\n{"foo": "bar"}\n]'
with open(file_path, 'r') as f:
self.assertEqual(f.read(), content)

View File

@ -7,8 +7,11 @@ import subprocess
import sys
import tempfile
from contextlib import contextmanager
from itertools import chain
from os.path import exists, join, abspath
from pathlib import Path
from shutil import rmtree, copytree
from stat import S_IWRITE as ANYONE_WRITE_PERMISSION
from tempfile import mkdtemp
from threading import Timer
from unittest import skipIf
@ -17,6 +20,7 @@ from twisted.trial import unittest
import scrapy
from scrapy.commands import ScrapyCommand
from scrapy.commands.startproject import IGNORE
from scrapy.settings import Settings
from scrapy.utils.python import to_unicode
from scrapy.utils.test import get_testenv
@ -121,6 +125,29 @@ class StartprojectTest(ProjectTest):
self.assertEqual(2, self.call('startproject', self.project_name, project_dir, 'another_params'))
def get_permissions_dict(path, renamings=None, ignore=None):
renamings = renamings or tuple()
permissions_dict = {
'.': os.stat(path).st_mode,
}
for root, dirs, files in os.walk(path):
nodes = list(chain(dirs, files))
if ignore:
ignored_names = ignore(root, nodes)
nodes = [node for node in nodes if node not in ignored_names]
for node in nodes:
absolute_path = os.path.join(root, node)
relative_path = os.path.relpath(absolute_path, path)
for search_string, replacement in renamings:
relative_path = relative_path.replace(
search_string,
replacement
)
permissions = os.stat(absolute_path).st_mode
permissions_dict[relative_path] = permissions
return permissions_dict
class StartprojectTemplatesTest(ProjectTest):
def setUp(self):
@ -141,6 +168,149 @@ class StartprojectTemplatesTest(ProjectTest):
self.assertIn(self.tmpl_proj, out)
assert exists(join(self.proj_path, 'root_template'))
def test_startproject_permissions_from_writable(self):
"""Check that generated files have the right permissions when the
template folder has the same permissions as in the project, i.e.
everything is writable."""
scrapy_path = scrapy.__path__[0]
project_template = os.path.join(scrapy_path, 'templates', 'project')
project_name = 'startproject1'
renamings = (
('module', project_name),
('.tmpl', ''),
)
expected_permissions = get_permissions_dict(
project_template,
renamings,
IGNORE,
)
destination = mkdtemp()
process = subprocess.Popen(
(
sys.executable,
'-m',
'scrapy.cmdline',
'startproject',
project_name,
),
cwd=destination,
env=self.env,
)
process.wait()
project_dir = os.path.join(destination, project_name)
actual_permissions = get_permissions_dict(project_dir)
self.assertEqual(actual_permissions, expected_permissions)
def test_startproject_permissions_from_read_only(self):
"""Check that generated files have the right permissions when the
template folder has been made read-only, which is something that some
systems do.
See https://github.com/scrapy/scrapy/pull/4604
"""
scrapy_path = scrapy.__path__[0]
templates_dir = os.path.join(scrapy_path, 'templates')
project_template = os.path.join(templates_dir, 'project')
project_name = 'startproject2'
renamings = (
('module', project_name),
('.tmpl', ''),
)
expected_permissions = get_permissions_dict(
project_template,
renamings,
IGNORE,
)
def _make_read_only(path):
current_permissions = os.stat(path).st_mode
os.chmod(path, current_permissions & ~ANYONE_WRITE_PERMISSION)
read_only_templates_dir = str(Path(mkdtemp()) / 'templates')
copytree(templates_dir, read_only_templates_dir)
for root, dirs, files in os.walk(read_only_templates_dir):
for node in chain(dirs, files):
_make_read_only(os.path.join(root, node))
destination = mkdtemp()
process = subprocess.Popen(
(
sys.executable,
'-m',
'scrapy.cmdline',
'startproject',
project_name,
'--set',
'TEMPLATES_DIR={}'.format(read_only_templates_dir),
),
cwd=destination,
env=self.env,
)
process.wait()
project_dir = os.path.join(destination, project_name)
actual_permissions = get_permissions_dict(project_dir)
self.assertEqual(actual_permissions, expected_permissions)
def test_startproject_permissions_unchanged_in_destination(self):
"""Check that pre-existing folders and files in the destination folder
do not see their permissions modified."""
scrapy_path = scrapy.__path__[0]
project_template = os.path.join(scrapy_path, 'templates', 'project')
project_name = 'startproject3'
renamings = (
('module', project_name),
('.tmpl', ''),
)
expected_permissions = get_permissions_dict(
project_template,
renamings,
IGNORE,
)
destination = mkdtemp()
project_dir = os.path.join(destination, project_name)
existing_nodes = {
oct(permissions)[2:] + extension: permissions
for extension in ('', '.d')
for permissions in (
0o444, 0o555, 0o644, 0o666, 0o755, 0o777,
)
}
os.mkdir(project_dir)
project_dir_path = Path(project_dir)
for node, permissions in existing_nodes.items():
path = project_dir_path / node
if node.endswith('.d'):
path.mkdir(mode=permissions)
else:
path.touch(mode=permissions)
expected_permissions[node] = path.stat().st_mode
process = subprocess.Popen(
(
sys.executable,
'-m',
'scrapy.cmdline',
'startproject',
project_name,
'.',
),
cwd=project_dir,
env=self.env,
)
process.wait()
actual_permissions = get_permissions_dict(project_dir)
self.assertEqual(actual_permissions, expected_permissions)
class CommandTest(ProjectTest):

View File

@ -5,8 +5,7 @@ from gzip import GzipFile
from scrapy.spiders import Spider
from scrapy.http import Response, Request, HtmlResponse
from scrapy.downloadermiddlewares.httpcompression import HttpCompressionMiddleware, \
ACCEPTED_ENCODINGS
from scrapy.downloadermiddlewares.httpcompression import HttpCompressionMiddleware, ACCEPTED_ENCODINGS
from scrapy.responsetypes import responsetypes
from scrapy.utils.gz import gunzip
from tests import tests_datadir

View File

@ -77,12 +77,9 @@ class RedirectMiddlewareTest(unittest.TestCase):
assert isinstance(req2, Request)
self.assertEqual(req2.url, url2)
self.assertEqual(req2.method, 'GET')
assert 'Content-Type' not in req2.headers, \
"Content-Type header must not be present in redirected request"
assert 'Content-Length' not in req2.headers, \
"Content-Length header must not be present in redirected request"
assert not req2.body, \
"Redirected body must be empty, not '%s'" % req2.body
assert 'Content-Type' not in req2.headers, "Content-Type header must not be present in redirected request"
assert 'Content-Length' not in req2.headers, "Content-Length header must not be present in redirected request"
assert not req2.body, "Redirected body must be empty, not '%s'" % req2.body
# response without Location header but with status code is 3XX should be ignored
del rsp.headers['Location']
@ -244,12 +241,9 @@ class MetaRefreshMiddlewareTest(unittest.TestCase):
assert isinstance(req2, Request)
self.assertEqual(req2.url, 'http://example.org/newpage')
self.assertEqual(req2.method, 'GET')
assert 'Content-Type' not in req2.headers, \
"Content-Type header must not be present in redirected request"
assert 'Content-Length' not in req2.headers, \
"Content-Length header must not be present in redirected request"
assert not req2.body, \
"Redirected body must be empty, not '%s'" % req2.body
assert 'Content-Type' not in req2.headers, "Content-Type header must not be present in redirected request"
assert 'Content-Length' not in req2.headers, "Content-Length header must not be present in redirected request"
assert not req2.body, "Redirected body must be empty, not '%s'" % req2.body
def test_max_redirect_times(self):
self.mw.max_redirect_times = 1

View File

@ -88,8 +88,7 @@ class SelectorTestCase(unittest.TestCase):
"""Check that classes are using slots and are weak-referenceable"""
x = Selector(text='')
weakref.ref(x)
assert not hasattr(x, '__dict__'), "%s does not use __slots__" % \
x.__class__.__name__
assert not hasattr(x, '__dict__'), "%s does not use __slots__" % x.__class__.__name__
def test_selector_bad_args(self):
with self.assertRaisesRegex(ValueError, 'received both response and text'):

View File

@ -86,8 +86,7 @@ class BaseSettingsTest(unittest.TestCase):
def test_set_calls_settings_attributes_methods_on_update(self):
attr = SettingsAttribute('value', 10)
with mock.patch.object(attr, '__setattr__') as mock_setattr, \
mock.patch.object(attr, 'set') as mock_set:
with mock.patch.object(attr, '__setattr__') as mock_setattr, mock.patch.object(attr, 'set') as mock_set:
self.settings.attributes = {'TEST_OPTION': attr}

View File

@ -11,8 +11,14 @@ from scrapy import signals
from scrapy.settings import Settings
from scrapy.http import Request, Response, TextResponse, XmlResponse, HtmlResponse
from scrapy.spiders.init import InitSpider
from scrapy.spiders import Spider, CrawlSpider, Rule, XMLFeedSpider, \
CSVFeedSpider, SitemapSpider
from scrapy.spiders import (
CSVFeedSpider,
CrawlSpider,
Rule,
SitemapSpider,
Spider,
XMLFeedSpider,
)
from scrapy.linkextractors import LinkExtractor
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.utils.test import get_crawler

View File

@ -6,16 +6,28 @@ from scrapy.http import Response, Request
from scrapy.settings import Settings
from scrapy.spiders import Spider
from scrapy.downloadermiddlewares.redirect import RedirectMiddleware
from scrapy.spidermiddlewares.referer import RefererMiddleware, \
POLICY_NO_REFERRER, POLICY_NO_REFERRER_WHEN_DOWNGRADE, \
POLICY_SAME_ORIGIN, POLICY_ORIGIN, POLICY_ORIGIN_WHEN_CROSS_ORIGIN, \
POLICY_SCRAPY_DEFAULT, POLICY_UNSAFE_URL, \
POLICY_STRICT_ORIGIN, POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN, \
DefaultReferrerPolicy, \
NoReferrerPolicy, NoReferrerWhenDowngradePolicy, \
OriginWhenCrossOriginPolicy, OriginPolicy, \
StrictOriginWhenCrossOriginPolicy, StrictOriginPolicy, \
SameOriginPolicy, UnsafeUrlPolicy, ReferrerPolicy
from scrapy.spidermiddlewares.referer import (
DefaultReferrerPolicy,
NoReferrerPolicy,
NoReferrerWhenDowngradePolicy,
OriginPolicy,
OriginWhenCrossOriginPolicy,
POLICY_NO_REFERRER,
POLICY_NO_REFERRER_WHEN_DOWNGRADE,
POLICY_ORIGIN,
POLICY_ORIGIN_WHEN_CROSS_ORIGIN,
POLICY_SAME_ORIGIN,
POLICY_SCRAPY_DEFAULT,
POLICY_STRICT_ORIGIN,
POLICY_STRICT_ORIGIN_WHEN_CROSS_ORIGIN,
POLICY_UNSAFE_URL,
RefererMiddleware,
ReferrerPolicy,
SameOriginPolicy,
StrictOriginPolicy,
StrictOriginWhenCrossOriginPolicy,
UnsafeUrlPolicy,
)
class TestRefererMiddleware(TestCase):

View File

@ -29,8 +29,7 @@ class CurlToRequestKwargsTest(unittest.TestCase):
self._test_command(curl_command, expected_result)
def test_get_basic_auth(self):
curl_command = 'curl "https://api.test.com/" -u ' \
'"some_username:some_password"'
curl_command = 'curl "https://api.test.com/" -u "some_username:some_password"'
expected_result = {
"method": "GET",
"url": "https://api.test.com/",
@ -212,8 +211,7 @@ class CurlToRequestKwargsTest(unittest.TestCase):
with warnings.catch_warnings(): # avoid warning when executing tests
warnings.simplefilter('ignore')
curl_command = 'curl --bar --baz http://www.example.com'
expected_result = \
{"method": "GET", "url": "http://www.example.com"}
expected_result = {"method": "GET", "url": "http://www.example.com"}
self.assertEqual(curl_to_request_kwargs(curl_command), expected_result)
# case 2: ignore_unknown_options=False (raise exception):

View File

@ -2,8 +2,13 @@ from twisted.trial import unittest
from twisted.internet import reactor, defer
from twisted.python.failure import Failure
from scrapy.utils.defer import mustbe_deferred, process_chain, \
process_chain_both, process_parallel, iter_errback
from scrapy.utils.defer import (
iter_errback,
mustbe_deferred,
process_chain,
process_chain_both,
process_parallel,
)
class MustbeDeferredTest(unittest.TestCase):

View File

@ -15,18 +15,20 @@ class XmliterTestCase(unittest.TestCase):
xmliter = staticmethod(xmliter)
def test_xmliter(self):
body = b"""<?xml version="1.0" encoding="UTF-8"?>\
body = b"""
<?xml version="1.0" encoding="UTF-8"?>
<products xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:noNamespaceSchemaLocation="someschmea.xsd">\
<product id="001">\
<type>Type 1</type>\
<name>Name 1</name>\
</product>\
<product id="002">\
<type>Type 2</type>\
<name>Name 2</name>\
</product>\
</products>"""
xsi:noNamespaceSchemaLocation="someschmea.xsd">
<product id="001">
<type>Type 1</type>
<name>Name 1</name>
</product>
<product id="002">
<type>Type 2</type>
<name>Name 2</name>
</product>
</products>
"""
response = XmlResponse(url="http://example.com", body=body)
attrs = []
@ -115,7 +117,7 @@ class XmliterTestCase(unittest.TestCase):
[[u'one'], [u'two']])
def test_xmliter_namespaces(self):
body = b"""\
body = b"""
<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns:g="http://base.google.com/ns/1.0">
<channel>
@ -185,7 +187,7 @@ class LxmlXmliterTestCase(XmliterTestCase):
xmliter = staticmethod(xmliter_lxml)
def test_xmliter_iterate_namespace(self):
body = b"""\
body = b"""
<?xml version="1.0" encoding="UTF-8"?>
<rss version="2.0" xmlns="http://base.google.com/ns/1.0">
<channel>
@ -214,7 +216,7 @@ class LxmlXmliterTestCase(XmliterTestCase):
self.assertEqual(node.xpath('text()').getall(), ['http://www.mydummycompany.com/images/item2.jpg'])
def test_xmliter_namespaces_prefix(self):
body = b"""\
body = b"""
<?xml version="1.0" encoding="UTF-8"?>
<root>
<h:table xmlns:h="http://www.w3.org/TR/html4/">

View File

@ -1,7 +1,11 @@
import unittest
from scrapy.http import Request
from scrapy.utils.request import request_fingerprint, _fingerprint_cache, \
request_authenticate, request_httprepr
from scrapy.utils.request import (
_fingerprint_cache,
request_authenticate,
request_fingerprint,
request_httprepr,
)
class UtilsRequestTest(unittest.TestCase):

View File

@ -37,8 +37,7 @@ class ResponseUtilsTest(unittest.TestCase):
self.assertIn(b'<base href="' + to_bytes(url) + b'">', bbody)
return True
response = HtmlResponse(url, body=body)
assert open_in_browser(response, _openfunc=browser_open), \
"Browser not called"
assert open_in_browser(response, _openfunc=browser_open), "Browser not called"
resp = Response(url, body=body)
self.assertRaises(TypeError, open_in_browser, resp, debug=True)

View File

@ -156,8 +156,7 @@ Disallow: /forum/active/
def test_sitemap_blanklines(self):
"""Assert we can deal with starting blank lines before <xml> tag"""
s = Sitemap(b"""\
s = Sitemap(b"""
<?xml version="1.0" encoding="UTF-8"?>
<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">

View File

@ -213,8 +213,7 @@ def create_guess_scheme_t(args):
def do_expected(self):
url = guess_scheme(args[0])
assert url.startswith(args[1]), \
'Wrong scheme guessed: for `%s` got `%s`, expected `%s...`' % (
args[0], url, args[1])
'Wrong scheme guessed: for `%s` got `%s`, expected `%s...`' % (args[0], url, args[1])
return do_expected

View File

@ -356,9 +356,8 @@ class WebClientTestCase(unittest.TestCase):
""" Test that non-standart body encoding matches
Content-Encoding header """
body = b'\xd0\x81\xd1\x8e\xd0\xaf'
return getPage(
self.getURL('encoding'), body=body, response_transform=lambda r: r)\
.addCallback(self._check_Encoding, body)
dfd = getPage(self.getURL('encoding'), body=body, response_transform=lambda r: r)
return dfd.addCallback(self._check_Encoding, body)
def _check_Encoding(self, response, original_body):
content_encoding = to_unicode(response.headers[b'Content-Encoding'])