Merge remote-tracking branch 'upstream/master' into ISSUE-4250-add_batch_deliveries

This commit is contained in:
BroodingKangaroo 2020-05-21 12:43:04 +03:00
commit 5f6c1dceb1
56 changed files with 557 additions and 427 deletions

View File

@ -1,4 +1,5 @@
version: 2
formats: all
sphinx:
configuration: docs/conf.py
fail_on_warning: true

View File

@ -17,12 +17,16 @@ matrix:
python: 3.7 # Keep in sync with .readthedocs.yml
- env: TOXENV=pypy3
- env: TOXENV=pinned
python: 3.5.1
dist: trusty
- env: TOXENV=asyncio
python: 3.5.1 # We use additional code to support 3.5.3 and earlier
dist: trusty
- env: TOXENV=py
python: 3.5
- env: TOXENV=pinned
python: 3.5
- env: TOXENV=asyncio
python: 3.5.2
python: 3.5 # We use specific code to support >= 3.5.4, < 3.6
- env: TOXENV=py
python: 3.6
- env: TOXENV=py

View File

@ -40,7 +40,7 @@ including a list of features.
Requirements
============
* Python 3.5+
* Python 3.5.1+
* Works on Linux, Windows, macOS, BSD
Install

View File

@ -100,6 +100,9 @@ exclude_trees = ['.build']
# The name of the Pygments (syntax highlighting) style to use.
pygments_style = 'sphinx'
# List of Sphinx warnings that will not be raised
suppress_warnings = ['epub.unknown_project_files']
# Options for HTML output
# -----------------------

View File

@ -69,7 +69,7 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars
What Python versions does Scrapy support?
-----------------------------------------
Scrapy is supported under Python 3.5+
Scrapy is supported under Python 3.5.1+
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
Python 3 support was added in Scrapy 1.1.
PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5.
@ -342,14 +342,14 @@ method for this purpose. For example::
from copy import deepcopy
from scrapy.item import BaseItem
from scrapy.item import Item
class MultiplyItemsMiddleware:
def process_spider_output(self, response, result, spider):
for item in result:
if isinstance(item, (BaseItem, dict)):
if isinstance(item, (Item, dict)):
for _ in range(item['multiply_by']):
yield deepcopy(item)

View File

@ -7,7 +7,7 @@ Installation guide
Installing Scrapy
=================
Scrapy runs on Python 3.5 or above under CPython (default Python
Scrapy runs on Python 3.5.1 or above under CPython (default Python
implementation) and PyPy (starting with PyPy 5.9).
If you're using `Anaconda`_ or `Miniconda`_, you can install the package from

View File

@ -257,6 +257,4 @@ Field objects
Other classes related to Item
=============================
.. autoclass:: BaseItem
.. autoclass:: ItemMeta

View File

@ -262,7 +262,6 @@ scrapy.utils.log module
This is an example on how to redirect ``INFO`` or higher messages to a file::
import logging
from scrapy.utils.log import configure_logging
logging.basicConfig(
filename='log.txt',

View File

@ -50,7 +50,7 @@ this:
4. When the files are downloaded, another field (``files``) will be populated
with the results. This field will contain a list of dicts with information
about the downloaded files, such as the downloaded path, the original
scraped url (taken from the ``file_urls`` field) , and the file checksum.
scraped url (taken from the ``file_urls`` field), the file checksum and the file status.
The files in the list of the ``files`` field will retain the same order of
the original ``file_urls`` field. If some file failed downloading, an
error will be logged and the file won't be present in the ``files`` field.
@ -470,6 +470,14 @@ See here the methods that you can override in your custom Files Pipeline:
* ``checksum`` - a `MD5 hash`_ of the image contents
* ``status`` - the file status indication. It can be one of the following:
* ``downloaded`` - file was downloaded.
* ``uptodate`` - file was not downloaded, as it was downloaded recently,
according to the file expiration policy.
* ``cached`` - file was already scheduled for download, by another item
sharing the same file.
The list of tuples received by :meth:`~item_completed` is
guaranteed to retain the same order of the requests returned from the
:meth:`~get_media_requests` method.
@ -479,7 +487,8 @@ See here the methods that you can override in your custom Files Pipeline:
[(True,
{'checksum': '2b00042f7481c7b056c4b410d28f33cf',
'path': 'full/0a79c461a4062ac383dc4fade7bc09f1384a3910.jpg',
'url': 'http://www.example.com/files/product1.pdf'}),
'url': 'http://www.example.com/files/product1.pdf',
'status': 'downloaded'}),
(False,
Failure(...))]

View File

@ -20,227 +20,25 @@ addopts =
twisted = 1
markers =
only_asyncio: marks tests as only enabled when --reactor=asyncio is passed
flake8-max-line-length = 119
flake8-ignore =
W503
# Files that are only meant to provide top-level imports are expected not
# to use any of their imports:
# Exclude files that are meant to provide top-level imports
# E402: Module level import not at top of file
# F401: Module imported but unused
scrapy/core/downloader/handlers/http.py F401
scrapy/http/__init__.py F401
scrapy/linkextractors/__init__.py E402 F401
scrapy/spiders/__init__.py E402 F401
# Issues pending a review:
# extras
extras/qps-bench-server.py E501
extras/qpsclient.py E501 E501
# scrapy/commands
scrapy/commands/__init__.py E501
scrapy/commands/check.py E501
scrapy/commands/crawl.py E501
scrapy/commands/edit.py E501
scrapy/commands/fetch.py E501
scrapy/commands/genspider.py E501
scrapy/commands/parse.py E501
scrapy/commands/runspider.py E501
scrapy/commands/settings.py E501
scrapy/commands/shell.py E501
scrapy/commands/startproject.py E501
scrapy/commands/version.py E501
# scrapy/contracts
scrapy/contracts/__init__.py E501
# scrapy/core
scrapy/core/engine.py E501
scrapy/core/scheduler.py E501
scrapy/core/scraper.py E501
scrapy/core/spidermw.py E501
scrapy/core/downloader/__init__.py E501
scrapy/core/downloader/contextfactory.py E501
scrapy/core/downloader/middleware.py E501
scrapy/core/downloader/tls.py E501
scrapy/core/downloader/webclient.py E501
scrapy/core/downloader/handlers/__init__.py E501
scrapy/core/downloader/handlers/ftp.py E501
scrapy/core/downloader/handlers/http10.py E501
scrapy/core/downloader/handlers/http11.py E501
scrapy/core/downloader/handlers/s3.py E501
# scrapy/downloadermiddlewares
scrapy/downloadermiddlewares/ajaxcrawl.py E501
scrapy/downloadermiddlewares/decompression.py E501
scrapy/downloadermiddlewares/defaultheaders.py E501
scrapy/downloadermiddlewares/httpcache.py E501
scrapy/downloadermiddlewares/httpcompression.py E501
scrapy/downloadermiddlewares/httpproxy.py E501
scrapy/downloadermiddlewares/redirect.py E501
scrapy/downloadermiddlewares/retry.py E501
scrapy/downloadermiddlewares/robotstxt.py E501
scrapy/downloadermiddlewares/stats.py E501
# scrapy/extensions
scrapy/extensions/closespider.py E501
scrapy/extensions/corestats.py E501
scrapy/extensions/feedexport.py E501
scrapy/extensions/httpcache.py E501
scrapy/extensions/memdebug.py E501
scrapy/extensions/spiderstate.py E501
scrapy/extensions/telnet.py E501
scrapy/extensions/throttle.py E501
# scrapy/http
scrapy/http/common.py E501
scrapy/http/cookies.py E501
scrapy/http/request/__init__.py E501
scrapy/http/request/form.py E501
scrapy/http/request/json_request.py E501
scrapy/http/response/__init__.py E501
scrapy/http/response/text.py E501
# scrapy/linkextractors
scrapy/linkextractors/__init__.py E501 E402
scrapy/linkextractors/lxmlhtml.py E501
# scrapy/loader
scrapy/loader/__init__.py E501
scrapy/loader/processors.py E501
# scrapy/pipelines
scrapy/pipelines/__init__.py E501
scrapy/pipelines/files.py E501
scrapy/pipelines/images.py E501
scrapy/pipelines/media.py E501
# scrapy/selector
scrapy/__init__.py E402
scrapy/selector/__init__.py F403
scrapy/selector/unified.py E501
# scrapy/settings
scrapy/settings/__init__.py E501
scrapy/settings/default_settings.py E501
scrapy/settings/deprecated.py E501
# scrapy/spidermiddlewares
scrapy/spidermiddlewares/httperror.py E501
scrapy/spidermiddlewares/offsite.py E501
scrapy/spidermiddlewares/referer.py E501
scrapy/spidermiddlewares/urllength.py E501
# scrapy/spiders
scrapy/spiders/__init__.py E501 E402
scrapy/spiders/crawl.py E501
scrapy/spiders/feed.py E501
scrapy/spiders/sitemap.py E501
# scrapy/utils
scrapy/utils/asyncio.py E501
scrapy/utils/benchserver.py E501
scrapy/utils/conf.py E402 E501
scrapy/utils/datatypes.py E501
scrapy/utils/decorators.py E501
scrapy/utils/defer.py E501
scrapy/utils/deprecate.py E501
scrapy/utils/gz.py E501
scrapy/spiders/__init__.py E402
scrapy/utils/http.py F403
scrapy/utils/httpobj.py E501
scrapy/utils/iterators.py E501
scrapy/utils/log.py E501
scrapy/utils/markup.py F403
scrapy/utils/misc.py E501
scrapy/utils/multipart.py F403
scrapy/utils/project.py E501
scrapy/utils/python.py E501
scrapy/utils/reactor.py E501
scrapy/utils/reqser.py E501
scrapy/utils/request.py E501
scrapy/utils/response.py E501
scrapy/utils/signal.py E501
scrapy/utils/sitemap.py E501
scrapy/utils/spider.py E501
scrapy/utils/ssl.py E501
scrapy/utils/test.py E501
scrapy/utils/url.py E501 F403 F405
# scrapy
scrapy/__init__.py E402 E501
scrapy/cmdline.py E501
scrapy/crawler.py E501
scrapy/dupefilters.py E501
scrapy/exceptions.py E501
scrapy/exporters.py E501
scrapy/interfaces.py E501
scrapy/item.py E501
scrapy/link.py E501
scrapy/logformatter.py E501
scrapy/mail.py E402 E501
scrapy/middleware.py E501
scrapy/pqueues.py E501
scrapy/resolver.py E501
scrapy/responsetypes.py E501
scrapy/robotstxt.py E501
scrapy/shell.py E501
scrapy/signalmanager.py E501
scrapy/spiderloader.py E501
scrapy/squeues.py E501
scrapy/statscollectors.py E501
# tests
tests/__init__.py E402 E501
tests/mockserver.py E501
tests/spiders.py E501
tests/test_closespider.py E501
tests/test_command_fetch.py E501
tests/test_command_parse.py E501
tests/test_command_shell.py E501
tests/test_commands.py E501
tests/test_contracts.py E501
tests/test_crawl.py E501
tests/test_crawler.py E501
tests/test_dependencies.py E501
tests/test_downloader_handlers.py E501
tests/test_downloadermiddleware.py E501
tests/test_downloadermiddleware_ajaxcrawlable.py E501
tests/test_downloadermiddleware_cookies.py E501
tests/test_downloadermiddleware_defaultheaders.py E501
tests/test_downloadermiddleware_downloadtimeout.py E501
tests/test_downloadermiddleware_httpcache.py E501
tests/test_downloadermiddleware_httpcompression.py E501
tests/test_downloadermiddleware_decompression.py E501
tests/test_downloadermiddleware_httpproxy.py E501
tests/test_downloadermiddleware_redirect.py E501
tests/test_downloadermiddleware_retry.py E501
tests/test_downloadermiddleware_robotstxt.py E501
tests/test_downloadermiddleware_stats.py E501
tests/test_dupefilters.py E501
tests/test_engine.py E501
tests/test_exporters.py E501
tests/test_feedexport.py E501
tests/test_http_cookies.py E501
tests/test_http_headers.py E501
tests/test_http_request.py E402 E501
tests/test_http_response.py E501
tests/test_item.py E501
tests/test_link.py E501
tests/test_linkextractors.py E501
tests/test_loader.py E501 E741
tests/test_logformatter.py E501
tests/test_mail.py E501
tests/test_middleware.py E501
tests/test_pipeline_crawl.py E501
tests/test_pipeline_files.py E501
tests/test_pipeline_images.py E501
tests/test_pipeline_media.py E501
tests/test_proxy_connect.py E501
tests/test_request_cb_kwargs.py E501
tests/test_responsetypes.py E501
tests/test_robotstxt_interface.py E501 E501
tests/test_scheduler.py E501
tests/test_selector.py E501
tests/test_spider.py E501
tests/test_spidermiddleware.py E501
tests/test_spidermiddleware_httperror.py E501
tests/test_spidermiddleware_offsite.py E501
tests/test_spidermiddleware_output_chain.py E501
tests/test_spidermiddleware_referer.py E501
tests/test_squeues.py E501
tests/test_utils_asyncio.py E501
tests/test_utils_conf.py E501
tests/test_utils_curl.py E501
tests/test_utils_datatypes.py E402 E501
tests/test_utils_defer.py E501
tests/test_utils_deprecate.py E501
tests/test_utils_http.py E501
tests/test_utils_iterators.py E501
tests/test_utils_python.py E501
tests/test_utils_reqser.py E501
tests/test_utils_request.py E501
tests/test_utils_response.py E501
tests/test_utils_sitemap.py E501
tests/test_utils_url.py E501 E501
tests/test_webclient.py E501 E402
tests/test_cmdline/__init__.py E501
tests/test_settings/__init__.py E501
tests/test_spiderloader/__init__.py E501
tests/test_utils_misc/__init__.py E501
scrapy/utils/url.py F403 F405
tests/test_loader.py E741
tests/test_webclient.py E402

View File

@ -165,6 +165,7 @@ if __name__ == '__main__':
try:
execute()
finally:
# Twisted prints errors in DebugInfo.__del__, but PyPy does not run gc.collect()
# on exit: http://doc.pypy.org/en/latest/cpython_differences.html?highlight=gc.collect#differences-related-to-garbage-collection-strategies
# Twisted prints errors in DebugInfo.__del__, but PyPy does not run gc.collect() on exit:
# http://doc.pypy.org/en/latest/cpython_differences.html
# ?highlight=gc.collect#differences-related-to-garbage-collection-strategies
garbage_collect()

View File

@ -5,7 +5,7 @@ from w3lib.url import is_url
from scrapy.commands import ScrapyCommand
from scrapy.http import Request
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
from scrapy.utils import display
from scrapy.utils.conf import arglist_to_dict
from scrapy.utils.spider import iterate_spider_output, spidercls_for_request
@ -117,7 +117,7 @@ class Command(ScrapyCommand):
items, requests = [], []
for x in iterate_spider_output(callback(response, **cb_kwargs)):
if isinstance(x, (BaseItem, dict)):
if isinstance(x, (_BaseItem, dict)):
items.append(x)
elif isinstance(x, Request):
requests.append(x)

View File

@ -1,6 +1,6 @@
import json
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
from scrapy.http import Request
from scrapy.exceptions import ContractFail
@ -51,8 +51,8 @@ class ReturnsContract(Contract):
objects = {
'request': Request,
'requests': Request,
'item': (BaseItem, dict),
'items': (BaseItem, dict),
'item': (_BaseItem, dict),
'items': (_BaseItem, dict),
}
def __init__(self, *args, **kwargs):
@ -103,7 +103,7 @@ class ScrapesContract(Contract):
def post_process(self, output):
for x in output:
if isinstance(x, (BaseItem, dict)):
if isinstance(x, (_BaseItem, dict)):
missing = [arg for arg in self.args if arg not in x]
if missing:
raise ContractFail(

View File

@ -14,7 +14,7 @@ from scrapy.utils.log import logformatter_adapter, failure_to_exc_info
from scrapy.exceptions import CloseSpider, DropItem, IgnoreRequest
from scrapy import signals
from scrapy.http import Request, Response
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
from scrapy.core.spidermw import SpiderMiddlewareManager
@ -191,7 +191,7 @@ class Scraper:
"""
if isinstance(output, Request):
self.crawler.engine.crawl(request=output, spider=spider)
elif isinstance(output, (BaseItem, dict)):
elif isinstance(output, (_BaseItem, dict)):
self.slot.itemproc_size += 1
dfd = self.itemproc.process_item(output, spider)
dfd.addBoth(self._itemproc_finished, output, response, spider)

View File

@ -12,7 +12,7 @@ from xml.sax.saxutils import XMLGenerator
from scrapy.utils.serialize import ScrapyJSONEncoder
from scrapy.utils.python import to_bytes, to_unicode, is_listlike
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
from scrapy.exceptions import ScrapyDeprecationWarning
@ -312,7 +312,7 @@ class PythonItemExporter(BaseItemExporter):
return serializer(value)
def _serialize_value(self, value):
if isinstance(value, BaseItem):
if isinstance(value, _BaseItem):
return self.export_item(value)
if isinstance(value, dict):
return dict(self._serialize_dict(value))

View File

@ -65,7 +65,7 @@ class TextResponse(Response):
"""Return body as unicode"""
warnings.warn('Response.body_as_unicode() is deprecated, '
'please use Response.text instead.',
ScrapyDeprecationWarning)
ScrapyDeprecationWarning, stacklevel=2)
return self.text
@property

View File

@ -14,28 +14,39 @@ from scrapy.utils.deprecate import ScrapyDeprecationWarning
from scrapy.utils.trackref import object_ref
class BaseItem(object_ref):
"""Base class for all scraped items.
In Scrapy, an object is considered an *item* if it is an instance of either
:class:`BaseItem` or :class:`dict`. For example, when the output of a
spider callback is evaluated, only instances of :class:`BaseItem` or
:class:`dict` are passed to :ref:`item pipelines <topics-item-pipeline>`.
If you need instances of a custom class to be considered items by Scrapy,
you must inherit from either :class:`BaseItem` or :class:`dict`.
Unlike instances of :class:`dict`, instances of :class:`BaseItem` may be
:ref:`tracked <topics-leaks-trackrefs>` to debug memory leaks.
class _BaseItem(object_ref):
"""
Temporary class used internally to avoid the deprecation
warning raised by isinstance checks using BaseItem.
"""
pass
class _BaseItemMeta(ABCMeta):
def __instancecheck__(cls, instance):
if cls is BaseItem:
warn('scrapy.item.BaseItem is deprecated, please use scrapy.item.Item instead',
ScrapyDeprecationWarning, stacklevel=2)
return super().__instancecheck__(instance)
class BaseItem(_BaseItem, metaclass=_BaseItemMeta):
"""
Deprecated, please use :class:`scrapy.item.Item` instead
"""
def __new__(cls, *args, **kwargs):
if issubclass(cls, BaseItem) and not (issubclass(cls, Item) or issubclass(cls, DictItem)):
warn('scrapy.item.BaseItem is deprecated, please use scrapy.item.Item instead',
ScrapyDeprecationWarning, stacklevel=2)
return super(BaseItem, cls).__new__(cls, *args, **kwargs)
class Field(dict):
"""Container of field metadata"""
class ItemMeta(ABCMeta):
class ItemMeta(_BaseItemMeta):
"""Metaclass_ of :class:`Item` that handles field definitions.
.. _metaclass: https://realpython.com/python-metaclasses
@ -68,8 +79,7 @@ class DictItem(MutableMapping, BaseItem):
def __new__(cls, *args, **kwargs):
if issubclass(cls, DictItem) and not issubclass(cls, Item):
warn('scrapy.item.DictItem is deprecated, please use '
'scrapy.item.Item instead',
warn('scrapy.item.DictItem is deprecated, please use scrapy.item.Item instead',
ScrapyDeprecationWarning, stacklevel=2)
return super(DictItem, cls).__new__(cls, *args, **kwargs)
@ -125,4 +135,24 @@ class DictItem(MutableMapping, BaseItem):
class Item(DictItem, metaclass=ItemMeta):
pass
"""
Base class for scraped items.
In Scrapy, an object is considered an ``item`` if it is an instance of either
:class:`Item` or :class:`dict`, or any subclass. For example, when the output of a
spider callback is evaluated, only instances of :class:`Item` or
:class:`dict` are passed to :ref:`item pipelines <topics-item-pipeline>`.
If you need instances of a custom class to be considered items by Scrapy,
you must inherit from either :class:`Item` or :class:`dict`.
Items must declare :class:`Field` attributes, which are processed and stored
in the ``fields`` attribute. This restricts the set of allowed field names
and prevents typos, raising ``KeyError`` when referring to undefined fields.
Additionally, fields can be used to define metadata and control the way
data is processed internally. Please refer to the :ref:`documentation
about fields <topics-items-fields>` for additional information.
Unlike instances of :class:`dict`, instances of :class:`Item` may be
:ref:`tracked <topics-leaks-trackrefs>` to debug memory leaks.
"""

View File

@ -432,7 +432,7 @@ class FilesPipeline(MediaPipeline):
self.inc_stats(info.spider, 'uptodate')
checksum = result.get('checksum', None)
return {'url': request.url, 'path': path, 'checksum': checksum}
return {'url': request.url, 'path': path, 'checksum': checksum, 'status': 'uptodate'}
path = self.file_path(request, info=info)
dfd = defer.maybeDeferred(self.store.stat_file, path, info)
@ -509,7 +509,7 @@ class FilesPipeline(MediaPipeline):
)
raise FileException(str(exc))
return {'url': request.url, 'path': path, 'checksum': checksum}
return {'url': request.url, 'path': path, 'checksum': checksum, 'status': status}
def inc_stats(self, spider, status):
spider.crawler.stats.inc_value('file_count', spider=spider)

View File

@ -17,10 +17,12 @@ def decode_robotstxt(robotstxt_body, spider, to_native_str_type=False):
except UnicodeDecodeError:
# If we found garbage or robots.txt in an encoding other than UTF-8, disregard it.
# Switch to 'allow all' state.
logger.warning("Failure while parsing robots.txt. "
"File either contains garbage or is in an encoding other than UTF-8, treating it as an empty file.",
exc_info=sys.exc_info(),
extra={'spider': spider})
logger.warning(
"Failure while parsing robots.txt. File either contains garbage or "
"is in an encoding other than UTF-8, treating it as an empty file.",
exc_info=sys.exc_info(),
extra={'spider': spider},
)
robotstxt_body = ''
return robotstxt_body

View File

@ -13,7 +13,7 @@ from w3lib.url import any_to_uri
from scrapy.crawler import Crawler
from scrapy.exceptions import IgnoreRequest
from scrapy.http import Request, Response
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
from scrapy.settings import Settings
from scrapy.spiders import Spider
from scrapy.utils.console import start_python_console
@ -26,8 +26,7 @@ from scrapy.utils.console import DEFAULT_PYTHON_SHELLS
class Shell:
relevant_classes = (Crawler, Spider, Request, Response, BaseItem,
Settings)
relevant_classes = (Crawler, Spider, Request, Response, _BaseItem, Settings)
def __init__(self, crawler, update_vars=None, code=None):
self.crawler = crawler

View File

@ -54,8 +54,12 @@ class Rule:
self.process_request = _get_method(self.process_request, spider)
self.process_request_argcount = len(get_func_args(self.process_request))
if self.process_request_argcount == 1:
msg = 'Rule.process_request should accept two arguments (request, response), accepting only one is deprecated'
warnings.warn(msg, category=ScrapyDeprecationWarning, stacklevel=2)
warnings.warn(
"Rule.process_request should accept two arguments "
"(request, response), accepting only one is deprecated",
category=ScrapyDeprecationWarning,
stacklevel=2,
)
def _process_request(self, request, response):
"""

View File

@ -52,7 +52,7 @@ class XMLFeedSpider(Spider):
"""This method is called for the nodes matching the provided tag name
(itertag). Receives the response and an Selector for each node.
Overriding this method is mandatory. Otherwise, you spider won't work.
This method must return either a BaseItem, a Request, or a list
This method must return either an item, a request, or a list
containing any of them.
"""

View File

@ -14,10 +14,10 @@ from w3lib.html import replace_entities
from scrapy.utils.datatypes import LocalWeakReferencedCache
from scrapy.utils.python import flatten, to_unicode
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
_ITERABLE_SINGLE_VALUES = dict, BaseItem, str, bytes
_ITERABLE_SINGLE_VALUES = dict, _BaseItem, str, bytes
def arg_to_iter(arg):

View File

@ -5,7 +5,7 @@ import decimal
from twisted.internet import defer
from scrapy.http import Request, Response
from scrapy.item import BaseItem
from scrapy.item import _BaseItem
class ScrapyJSONEncoder(json.JSONEncoder):
@ -26,7 +26,7 @@ class ScrapyJSONEncoder(json.JSONEncoder):
return str(o)
elif isinstance(o, defer.Deferred):
return str(o)
elif isinstance(o, BaseItem):
elif isinstance(o, _BaseItem):
return dict(o)
elif isinstance(o, Request):
return "<%s %s %s>" % (type(o).__name__, o.method, o.url)

View File

@ -56,7 +56,9 @@ class ShellTest(ProcessTest, SiteTest, unittest.TestCase):
@defer.inlineCallbacks
def test_redirect_not_follow_302(self):
_, out, _ = yield self.execute(['--no-redirect', self.url('/redirect-no-meta-refresh'), '-c', 'response.status'])
_, out, _ = yield self.execute(
['--no-redirect', self.url('/redirect-no-meta-refresh'), '-c', 'response.status']
)
assert out.strip().endswith(b'302')
@defer.inlineCallbacks

View File

@ -305,8 +305,10 @@ class CrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
def test_ipv6_default_name_resolver(self):
log = self.run_script('default_name_resolver.py')
self.assertIn('Spider closed (finished)', log)
self.assertIn("twisted.internet.error.DNSLookupError: DNS lookup failed: no results for hostname lookup: ::1.", log)
self.assertIn("'downloader/exception_type_count/twisted.internet.error.DNSLookupError': 1,", log)
self.assertIn(
"twisted.internet.error.DNSLookupError: DNS lookup failed: no results for hostname lookup: ::1.",
log)
def test_ipv6_alternative_name_resolver(self):
log = self.run_script('alternative_name_resolver.py')

View File

@ -493,7 +493,10 @@ class Http11TestCase(HttpTestCase):
class Https11TestCase(Http11TestCase):
scheme = 'https'
tls_log_message = 'SSL connection certificate: issuer "/C=IE/O=Scrapy/CN=localhost", subject "/C=IE/O=Scrapy/CN=localhost"'
tls_log_message = (
'SSL connection certificate: issuer "/C=IE/O=Scrapy/CN=localhost", '
'subject "/C=IE/O=Scrapy/CN=localhost"'
)
@defer.inlineCallbacks
def test_tls_logging(self):
@ -542,7 +545,10 @@ class Https11InvalidDNSPattern(Https11TestCase):
from service_identity.exceptions import CertificateError # noqa: F401
except ImportError:
raise unittest.SkipTest("cryptography lib is too old")
self.tls_log_message = 'SSL connection certificate: issuer "/C=IE/O=Scrapy/CN=127.0.0.1", subject "/C=IE/O=Scrapy/CN=127.0.0.1"'
self.tls_log_message = (
'SSL connection certificate: issuer "/C=IE/O=Scrapy/CN=127.0.0.1", '
'subject "/C=IE/O=Scrapy/CN=127.0.0.1"'
)
super(Https11InvalidDNSPattern, self).setUp()

View File

@ -124,7 +124,8 @@ class HttpCompressionTest(TestCase):
'Content-Encoding': 'gzip',
}
f = BytesIO()
plainbody = b"""<html><head><title>Some page</title><meta http-equiv="Content-Type" content="text/html; charset=gb2312">"""
plainbody = (b'<html><head><title>Some page</title>'
b'<meta http-equiv="Content-Type" content="text/html; charset=gb2312">')
zf = GzipFile(fileobj=f, mode='wb')
zf.write(plainbody)
zf.close()
@ -142,7 +143,8 @@ class HttpCompressionTest(TestCase):
'Content-Encoding': 'gzip',
}
f = BytesIO()
plainbody = b"""<html><head><title>Some page</title><meta http-equiv="Content-Type" content="text/html; charset=gb2312">"""
plainbody = (b'<html><head><title>Some page</title>'
b'<meta http-equiv="Content-Type" content="text/html; charset=gb2312">')
zf = GzipFile(fileobj=f, mode='wb')
zf.write(plainbody)
zf.close()
@ -158,7 +160,8 @@ class HttpCompressionTest(TestCase):
headers = {
'Content-Encoding': 'identity',
}
plainbody = b"""<html><head><title>Some page</title><meta http-equiv="Content-Type" content="text/html; charset=gb2312">"""
plainbody = (b'<html><head><title>Some page</title>'
b'<meta http-equiv="Content-Type" content="text/html; charset=gb2312">')
respcls = responsetypes.from_args(url="http://www.example.com/index", headers=headers, body=plainbody)
response = respcls("http://www.example.com/index", headers=headers, body=plainbody)
request = Request("http://www.example.com/index")

View File

@ -149,7 +149,10 @@ class RedirectMiddlewareTest(unittest.TestCase):
self.assertEqual(req2.url, 'http://scrapytest.org/redirected')
self.assertEqual(req2.meta['redirect_urls'], ['http://scrapytest.org/first'])
self.assertEqual(req3.url, 'http://scrapytest.org/redirected2')
self.assertEqual(req3.meta['redirect_urls'], ['http://scrapytest.org/first', 'http://scrapytest.org/redirected'])
self.assertEqual(
req3.meta['redirect_urls'],
['http://scrapytest.org/first', 'http://scrapytest.org/redirected']
)
def test_redirect_reasons(self):
req1 = Request('http://scrapytest.org/first')
@ -279,7 +282,10 @@ class MetaRefreshMiddlewareTest(unittest.TestCase):
self.assertEqual(req2.url, 'http://scrapytest.org/redirected')
self.assertEqual(req2.meta['redirect_urls'], ['http://scrapytest.org/first'])
self.assertEqual(req3.url, 'http://scrapytest.org/redirected2')
self.assertEqual(req3.meta['redirect_urls'], ['http://scrapytest.org/first', 'http://scrapytest.org/redirected'])
self.assertEqual(
req3.meta['redirect_urls'],
['http://scrapytest.org/first', 'http://scrapytest.org/redirected']
)
def test_redirect_reasons(self):
req1 = Request('http://scrapytest.org/first')

View File

@ -181,7 +181,8 @@ class RFPDupeFilterTest(unittest.TestCase):
(
'scrapy.dupefilters',
'DEBUG',
'Filtered duplicate request: <GET http://scrapytest.org/index.html> - no more duplicates will be shown (see DUPEFILTER_DEBUG to show all duplicates)'
'Filtered duplicate request: <GET http://scrapytest.org/index.html> - no more'
' duplicates will be shown (see DUPEFILTER_DEBUG to show all duplicates)'
)
)
@ -217,7 +218,8 @@ class RFPDupeFilterTest(unittest.TestCase):
(
'scrapy.dupefilters',
'DEBUG',
'Filtered duplicate request: <GET http://scrapytest.org/index.html> (referer: http://scrapytest.org/INDEX.html)'
'Filtered duplicate request: <GET http://scrapytest.org/index.html>'
' (referer: http://scrapytest.org/INDEX.html)'
)
)

View File

@ -218,8 +218,8 @@ class EngineTest(unittest.TestCase):
def _assert_visited_urls(self):
must_be_visited = ["/", "/redirect", "/redirected",
"/item1.html", "/item2.html", "/item999.html"]
urls_visited = set([rp[0].url for rp in self.run.respplug])
urls_expected = set([self.run.geturl(p) for p in must_be_visited])
urls_visited = {rp[0].url for rp in self.run.respplug}
urls_expected = {self.run.geturl(p) for p in must_be_visited}
assert urls_expected <= urls_visited, "URLs not visited: %s" % list(urls_expected - urls_visited)
def _assert_scheduled_requests(self, urls_to_visit=None):
@ -227,8 +227,8 @@ class EngineTest(unittest.TestCase):
paths_expected = ['/item999.html', '/item2.html', '/item1.html']
urls_requested = set([rq[0].url for rq in self.run.reqplug])
urls_expected = set([self.run.geturl(p) for p in paths_expected])
urls_requested = {rq[0].url for rq in self.run.reqplug}
urls_expected = {self.run.geturl(p) for p in paths_expected}
assert urls_expected <= urls_requested
scheduled_requests_count = len(self.run.reqplug)
dropped_requests_count = len(self.run.reqdropped)

View File

@ -111,7 +111,10 @@ class PythonItemExporterTest(BaseItemExporterTest):
ie = self._get_exporter()
exported = ie.export_item(i3)
self.assertEqual(type(exported), dict)
self.assertEqual(exported, {'age': {'age': {'age': '22', 'name': u'Joseph'}, 'name': u'Maria'}, 'name': 'Jesus'})
self.assertEqual(
exported,
{'age': {'age': {'age': '22', 'name': u'Joseph'}, 'name': u'Maria'}, 'name': 'Jesus'}
)
self.assertEqual(type(exported['age']), dict)
self.assertEqual(type(exported['age']['age']), dict)
@ -121,7 +124,10 @@ class PythonItemExporterTest(BaseItemExporterTest):
i3 = TestItem(name=u'Jesus', age=[i2])
ie = self._get_exporter()
exported = ie.export_item(i3)
self.assertEqual(exported, {'age': [{'age': [{'age': '22', 'name': u'Joseph'}], 'name': u'Maria'}], 'name': 'Jesus'})
self.assertEqual(
exported,
{'age': [{'age': [{'age': '22', 'name': u'Joseph'}], 'name': u'Maria'}], 'name': 'Jesus'}
)
self.assertEqual(type(exported['age'][0]), dict)
self.assertEqual(type(exported['age'][0]['age'][0]), dict)
@ -131,7 +137,10 @@ class PythonItemExporterTest(BaseItemExporterTest):
i3 = TestItem(name=u'Jesus', age=[i2])
ie = self._get_exporter()
exported = ie.export_item(i3)
self.assertEqual(exported, {'age': [{'age': [{'age': '22', 'name': u'Joseph'}], 'name': u'Maria'}], 'name': 'Jesus'})
self.assertEqual(
exported,
{'age': [{'age': [{'age': '22', 'name': u'Joseph'}], 'name': u'Maria'}], 'name': 'Jesus'}
)
self.assertEqual(type(exported['age'][0]), dict)
self.assertEqual(type(exported['age'][0]['age'][0]), dict)
@ -328,13 +337,19 @@ class XmlItemExporterTest(BaseItemExporterTest):
self.assertXmlEquivalent(fp.getvalue(), expected_value)
def _check_output(self):
expected_value = b'<?xml version="1.0" encoding="utf-8"?>\n<items><item><age>22</age><name>John\xc2\xa3</name></item></items>'
expected_value = (
b'<?xml version="1.0" encoding="utf-8"?>\n'
b'<items><item><age>22</age><name>John\xc2\xa3</name></item></items>'
)
self.assertXmlEquivalent(self.output.getvalue(), expected_value)
def test_multivalued_fields(self):
self.assertExportResult(
TestItem(name=[u'John\xa3', u'Doe']),
b'<?xml version="1.0" encoding="utf-8"?>\n<items><item><name><value>John\xc2\xa3</value><value>Doe</value></name></item></items>'
(
b'<?xml version="1.0" encoding="utf-8"?>\n'
b'<items><item><name><value>John\xc2\xa3</value><value>Doe</value></name></item></items>'
)
)
def test_nested_item(self):

View File

@ -732,10 +732,13 @@ class FeedExportTest(FeedExportTestBase):
items = [dict({'foo': u'Test\xd6'})]
formats = {
'json': u'[{"foo": "Test\\u00d6"}]'.encode('utf-8'),
'jsonlines': u'{"foo": "Test\\u00d6"}\n'.encode('utf-8'),
'xml': u'<?xml version="1.0" encoding="utf-8"?>\n<items><item><foo>Test\xd6</foo></item></items>'.encode('utf-8'),
'csv': u'foo\r\nTest\xd6\r\n'.encode('utf-8'),
'json': '[{"foo": "Test\\u00d6"}]'.encode('utf-8'),
'jsonlines': '{"foo": "Test\\u00d6"}\n'.encode('utf-8'),
'xml': (
'<?xml version="1.0" encoding="utf-8"?>\n'
'<items><item><foo>Test\xd6</foo></item></items>'
).encode('utf-8'),
'csv': 'foo\r\nTest\xd6\r\n'.encode('utf-8'),
}
for fmt, expected in formats.items():
@ -749,10 +752,13 @@ class FeedExportTest(FeedExportTestBase):
self.assertEqual(expected, data[fmt])
formats = {
'json': u'[{"foo": "Test\xd6"}]'.encode('latin-1'),
'jsonlines': u'{"foo": "Test\xd6"}\n'.encode('latin-1'),
'xml': u'<?xml version="1.0" encoding="latin-1"?>\n<items><item><foo>Test\xd6</foo></item></items>'.encode('latin-1'),
'csv': u'foo\r\nTest\xd6\r\n'.encode('latin-1'),
'json': '[{"foo": "Test\xd6"}]'.encode('latin-1'),
'jsonlines': '{"foo": "Test\xd6"}\n'.encode('latin-1'),
'xml': (
'<?xml version="1.0" encoding="latin-1"?>\n'
'<items><item><foo>Test\xd6</foo></item></items>'
).encode('latin-1'),
'csv': 'foo\r\nTest\xd6\r\n'.encode('latin-1'),
}
for fmt, expected in formats.items():
@ -771,9 +777,12 @@ class FeedExportTest(FeedExportTestBase):
items = [dict({'foo': u'FOO', 'bar': u'BAR'})]
formats = {
'json': u'[\n{"bar": "BAR"}\n]'.encode('utf-8'),
'xml': u'<?xml version="1.0" encoding="latin-1"?>\n<items>\n <item>\n <foo>FOO</foo>\n </item>\n</items>'.encode('latin-1'),
'csv': u'bar,foo\r\nBAR,FOO\r\n'.encode('utf-8'),
'json': '[\n{"bar": "BAR"}\n]'.encode('utf-8'),
'xml': (
'<?xml version="1.0" encoding="latin-1"?>\n'
'<items>\n <item>\n <foo>FOO</foo>\n </item>\n</items>'
).encode('latin-1'),
'csv': 'bar,foo\r\nBAR,FOO\r\n'.encode('utf-8'),
}
settings = {

View File

@ -549,8 +549,8 @@ class FormRequestTest(RequestTest):
self.assertEqual(urlparse(r1.url).hostname, "www.example.com")
self.assertEqual(urlparse(r1.url).path, "/this/get.php")
fs = _qs(r1)
self.assertEqual(set(fs[b'test']), set([b'val1', b'val2']))
self.assertEqual(set(fs[b'one']), set([b'two', b'three']))
self.assertEqual(set(fs[b'test']), {b'val1', b'val2'})
self.assertEqual(set(fs[b'one']), {b'two', b'three'})
self.assertEqual(fs[b'test2'], [b'xxx'])
self.assertEqual(fs[b'six'], [b'seven'])
@ -1047,7 +1047,7 @@ class FormRequestTest(RequestTest):
</form>''')
req = self.request_class.from_response(res)
fs = _qs(req)
self.assertEqual(set(fs), set([b'h2', b'i2', b'i1', b'i3', b'h1', b'i5', b'i4']))
self.assertEqual(set(fs), {b'h2', b'i2', b'i1', b'i3', b'h1', b'i5', b'i4'})
def test_from_response_xpath(self):
response = _buildresponse(
@ -1258,7 +1258,10 @@ class XmlRpcRequestTest(RequestTest):
class JsonRequestTest(RequestTest):
request_class = JsonRequest
default_method = 'GET'
default_headers = {b'Content-Type': [b'application/json'], b'Accept': [b'application/json, text/javascript, */*; q=0.01']}
default_headers = {
b'Content-Type': [b'application/json'],
b'Accept': [b'application/json, text/javascript, */*; q=0.01'],
}
def setUp(self):
warnings.simplefilter("always")

View File

@ -26,7 +26,11 @@ class BaseResponseTest(unittest.TestCase):
self.assertTrue(isinstance(self.response_class('http://example.com/', body=b''), self.response_class))
self.assertTrue(isinstance(self.response_class('http://example.com/', body=b'body'), self.response_class))
# test presence of all optional parameters
self.assertTrue(isinstance(self.response_class('http://example.com/', body=b'', headers={}, status=200), self.response_class))
self.assertTrue(
isinstance(
self.response_class('http://example.com/', body=b'', headers={}, status=200), self.response_class
)
)
r = self.response_class("http://www.example.com")
assert isinstance(r.url, str)
@ -324,13 +328,16 @@ class TextResponseTest(BaseResponseTest):
self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3'))
resp = self.response_class(url=u"http://www.example.com/price/\xa3", encoding='latin-1')
self.assertEqual(resp.url, 'http://www.example.com/price/\xa3')
resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=utf-8"]})
resp = self.response_class(u"http://www.example.com/price/\xa3",
headers={"Content-type": ["text/html; charset=utf-8"]})
self.assertEqual(resp.url, to_unicode(b'http://www.example.com/price/\xc2\xa3'))
resp = self.response_class(u"http://www.example.com/price/\xa3", headers={"Content-type": ["text/html; charset=iso-8859-1"]})
resp = self.response_class(u"http://www.example.com/price/\xa3",
headers={"Content-type": ["text/html; charset=iso-8859-1"]})
self.assertEqual(resp.url, 'http://www.example.com/price/\xa3')
def test_unicode_body(self):
unicode_string = u'\u043a\u0438\u0440\u0438\u043b\u043b\u0438\u0447\u0435\u0441\u043a\u0438\u0439 \u0442\u0435\u043a\u0441\u0442'
unicode_string = ('\u043a\u0438\u0440\u0438\u043b\u043b\u0438\u0447\u0435\u0441\u043a\u0438\u0439 '
'\u0442\u0435\u043a\u0441\u0442')
self.assertRaises(TypeError, self.response_class, 'http://www.example.com', body=u'unicode body')
original_string = unicode_string.encode('cp1251')
@ -345,13 +352,18 @@ class TextResponseTest(BaseResponseTest):
self.assertEqual(r1.text, unicode_string)
def test_encoding(self):
r1 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=utf-8"]}, body=b"\xc2\xa3")
r1 = self.response_class("http://www.example.com", body=b"\xc2\xa3",
headers={"Content-type": ["text/html; charset=utf-8"]})
r2 = self.response_class("http://www.example.com", encoding='utf-8', body=u"\xa3")
r3 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=iso-8859-1"]}, body=b"\xa3")
r3 = self.response_class("http://www.example.com", body=b"\xa3",
headers={"Content-type": ["text/html; charset=iso-8859-1"]})
r4 = self.response_class("http://www.example.com", body=b"\xa2\xa3")
r5 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=None"]}, body=b"\xc2\xa3")
r6 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=gb2312"]}, body=b"\xa8D")
r7 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=gbk"]}, body=b"\xa8D")
r5 = self.response_class("http://www.example.com", body=b"\xc2\xa3",
headers={"Content-type": ["text/html; charset=None"]})
r6 = self.response_class("http://www.example.com", body=b"\xa8D",
headers={"Content-type": ["text/html; charset=gb2312"]})
r7 = self.response_class("http://www.example.com", body=b"\xa8D",
headers={"Content-type": ["text/html; charset=gbk"]})
self.assertEqual(r1._headers_encoding(), "utf-8")
self.assertEqual(r2._headers_encoding(), None)
@ -697,7 +709,8 @@ class HtmlResponseTest(TextResponseTest):
body = b"""<html><head><title>Some page</title><meta http-equiv="Content-Type" content="text/html; charset=utf-8">
</head><body>Price: \xa3100</body></html>'
"""
r3 = self.response_class("http://www.example.com", headers={"Content-type": ["text/html; charset=iso-8859-1"]}, body=body)
r3 = self.response_class("http://www.example.com", body=body,
headers={"Content-type": ["text/html; charset=iso-8859-1"]})
self._assert_response_values(r3, 'iso-8859-1', body)
# make sure replace() preserves the encoding of the original response

View File

@ -4,7 +4,7 @@ from unittest import mock
from warnings import catch_warnings
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.item import ABCMeta, DictItem, Field, Item, ItemMeta
from scrapy.item import ABCMeta, _BaseItem, BaseItem, DictItem, Field, Item, ItemMeta
PY36_PLUS = (sys.version_info.major >= 3) and (sys.version_info.minor >= 6)
@ -131,12 +131,12 @@ class ItemTest(unittest.TestCase):
self.assertSortedEqual(list(item.values()), [u'New'])
def test_metaclass_inheritance(self):
class BaseItem(Item):
class ParentItem(Item):
name = Field()
keys = Field()
values = Field()
class TestItem(BaseItem):
class TestItem(ParentItem):
keys = Field()
i = TestItem()
@ -330,5 +330,77 @@ class DictItemTest(unittest.TestCase):
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
class BaseItemTest(unittest.TestCase):
def test_isinstance_check(self):
class SubclassedBaseItem(BaseItem):
pass
class SubclassedItem(Item):
pass
self.assertTrue(isinstance(BaseItem(), BaseItem))
self.assertTrue(isinstance(SubclassedBaseItem(), BaseItem))
self.assertTrue(isinstance(Item(), BaseItem))
self.assertTrue(isinstance(SubclassedItem(), BaseItem))
# make sure internal checks using private _BaseItem class succeed
self.assertTrue(isinstance(BaseItem(), _BaseItem))
self.assertTrue(isinstance(SubclassedBaseItem(), _BaseItem))
self.assertTrue(isinstance(Item(), _BaseItem))
self.assertTrue(isinstance(SubclassedItem(), _BaseItem))
def test_deprecation_warning(self):
"""
Make sure deprecation warnings are logged whenever BaseItem is used,
either instantiated or in an isinstance check
"""
with catch_warnings(record=True) as warnings:
BaseItem()
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
with catch_warnings(record=True) as warnings:
class SubclassedBaseItem(BaseItem):
pass
SubclassedBaseItem()
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
with catch_warnings(record=True) as warnings:
self.assertFalse(isinstance("foo", BaseItem))
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
with catch_warnings(record=True) as warnings:
self.assertTrue(isinstance(BaseItem(), BaseItem))
self.assertEqual(len(warnings), 1)
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
class ItemNoDeprecationWarningTest(unittest.TestCase):
def test_no_deprecation_warning(self):
"""
Make sure deprecation warnings are NOT logged whenever BaseItem subclasses are used.
"""
class SubclassedItem(Item):
pass
with catch_warnings(record=True) as warnings:
Item()
SubclassedItem()
_BaseItem()
self.assertFalse(isinstance("foo", _BaseItem))
self.assertFalse(isinstance("foo", Item))
self.assertFalse(isinstance("foo", SubclassedItem))
self.assertTrue(isinstance(_BaseItem(), _BaseItem))
self.assertTrue(isinstance(Item(), Item))
self.assertTrue(isinstance(SubclassedItem(), SubclassedItem))
self.assertEqual(len(warnings), 0)
if __name__ == "__main__":
unittest.main()

View File

@ -280,8 +280,8 @@ class Base:
def test_process_value(self):
"""Test restrict_xpaths with encodings"""
html = b"""
<a href="javascript:goToPage('../other/page.html','photo','width=600,height=540,scrollbars'); return false">Link text</a>
<a href="/about.html">About us</a>
<a href="javascript:goToPage('../other/page.html','photo','width=600,height=540,scrollbars'); return false">Text</a>
<a href="/about.html">About us</a>
"""
response = HtmlResponse("http://example.org/somepage/index.html", body=html, encoding='windows-1252')
@ -292,7 +292,7 @@ class Base:
lx = self.extractor_cls(process_value=process_value)
self.assertEqual(lx.extract_links(response),
[Link(url='http://example.org/other/page.html', text='Link text')])
[Link(url='http://example.org/other/page.html', text='Text')])
def test_base_url_with_restrict_xpaths(self):
html = b"""<html><head><title>Page title<title><base href="http://otherdomain.com/base/" />
@ -333,7 +333,10 @@ class Base:
self.assertEqual(lx.extract_links(self.response), [])
def test_tags(self):
html = b"""<html><area href="sample1.html"></area><a href="sample2.html">sample 2</a><img src="sample2.jpg"/></html>"""
html = (
b'<html><area href="sample1.html"></area>'
b'<a href="sample2.html">sample 2</a><img src="sample2.jpg"/></html>'
)
response = HtmlResponse("http://example.com/index.html", body=html)
lx = self.extractor_cls(tags=None)
@ -419,8 +422,10 @@ class Base:
[
Link(url='http://example.com/about.html', text=u'About us', fragment='', nofollow=False),
Link(url='http://example.com/follow.html', text=u'Follow this link', fragment='', nofollow=False),
Link(url='http://example.com/nofollow.html', text=u'Dont follow this one', fragment='', nofollow=True),
Link(url='http://example.com/nofollow2.html', text=u'Choose to follow or not', fragment='', nofollow=False),
Link(url='http://example.com/nofollow.html', text=u'Dont follow this one',
fragment='', nofollow=True),
Link(url='http://example.com/nofollow2.html', text=u'Choose to follow or not',
fragment='', nofollow=False),
Link(url='http://google.com/something', text=u'External link not to follow', nofollow=True),
]
)
@ -433,8 +438,10 @@ class Base:
[
Link(url='http://example.com/about.html', text=u'About us', fragment='', nofollow=False),
Link(url='http://example.com/follow.html', text=u'Follow this link', fragment='', nofollow=False),
Link(url='http://example.com/nofollow.html', text=u'Dont follow this one', fragment='', nofollow=True),
Link(url='http://example.com/nofollow2.html', text=u'Choose to follow or not', fragment='', nofollow=False),
Link(url='http://example.com/nofollow.html', text=u'Dont follow this one',
fragment='', nofollow=True),
Link(url='http://example.com/nofollow2.html', text=u'Choose to follow or not',
fragment='', nofollow=False),
Link(url='http://google.com/something', text=u'External link not to follow', nofollow=True),
]
)

View File

@ -601,7 +601,7 @@ class NoInputReprocessingItemLoader(BaseNoInputReprocessingLoader):
class NoInputReprocessingFromItemTest(unittest.TestCase):
"""
Loaders initialized from loaded items must not reprocess fields (BaseItem instances)
Loaders initialized from loaded items must not reprocess fields (Item instances)
"""
def test_avoid_reprocessing_with_initial_values_single(self):
il = NoInputReprocessingItemLoader(item=NoInputReprocessingItem(title='foo'))

View File

@ -51,10 +51,10 @@ class FileDownloadCrawlTestCase(TestCase):
store_setting_key = 'FILES_STORE'
media_key = 'files'
media_urls_key = 'file_urls'
expected_checksums = set([
expected_checksums = {
'5547178b89448faf0015a13f904c936e',
'c2281c83670e31d8aaab7cb642b824db',
'ed3f6538dc15d4d9179dae57319edc5f'])
'ed3f6538dc15d4d9179dae57319edc5f'}
def setUp(self):
self.mockserver = MockServer()
@ -91,6 +91,11 @@ class FileDownloadCrawlTestCase(TestCase):
file_dl_success = 'File (downloaded): Downloaded file from'
self.assertEqual(logs.count(file_dl_success), 3)
# check that the images/files status is `downloaded`
for item in items:
for i in item[self.media_key]:
self.assertEqual(i['status'], 'downloaded')
# check that the images/files checksums are what we know they should be
if self.expected_checksums is not None:
checksums = set(

View File

@ -38,27 +38,36 @@ class FilesPipelineTestCase(unittest.TestCase):
def test_file_path(self):
file_path = self.pipeline.file_path
self.assertEqual(file_path(Request("https://dev.mydeco.com/mydeco.pdf")),
'full/c9b564df929f4bc635bdd19fde4f3d4847c757c5.pdf')
self.assertEqual(file_path(Request("http://www.maddiebrown.co.uk///catalogue-items//image_54642_12175_95307.txt")),
'full/4ce274dd83db0368bafd7e406f382ae088e39219.txt')
self.assertEqual(file_path(Request("https://dev.mydeco.com/two/dirs/with%20spaces%2Bsigns.doc")),
'full/94ccc495a17b9ac5d40e3eabf3afcb8c2c9b9e1a.doc')
self.assertEqual(file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg")),
'full/4507be485f38b0da8a0be9eb2e1dfab8a19223f2.jpg')
self.assertEqual(file_path(Request("http://www.dorma.co.uk/images/product_details/2532/")),
'full/97ee6f8a46cbbb418ea91502fd24176865cf39b2')
self.assertEqual(file_path(Request("http://www.dorma.co.uk/images/product_details/2532")),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1')
self.assertEqual(file_path(Request("http://www.dorma.co.uk/images/product_details/2532"),
response=Response("http://www.dorma.co.uk/images/product_details/2532"),
info=object()),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1')
self.assertEqual(file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg.bohaha")),
'full/76c00cef2ef669ae65052661f68d451162829507')
self.assertEqual(file_path(Request("data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAR0AAACxCAMAAADOHZloAAACClBMVEX/\
self.assertEqual(
file_path(Request("https://dev.mydeco.com/mydeco.pdf")),
'full/c9b564df929f4bc635bdd19fde4f3d4847c757c5.pdf')
self.assertEqual(
file_path(Request("http://www.maddiebrown.co.uk///catalogue-items//image_54642_12175_95307.txt")),
'full/4ce274dd83db0368bafd7e406f382ae088e39219.txt')
self.assertEqual(
file_path(Request("https://dev.mydeco.com/two/dirs/with%20spaces%2Bsigns.doc")),
'full/94ccc495a17b9ac5d40e3eabf3afcb8c2c9b9e1a.doc')
self.assertEqual(
file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg")),
'full/4507be485f38b0da8a0be9eb2e1dfab8a19223f2.jpg')
self.assertEqual(
file_path(Request("http://www.dorma.co.uk/images/product_details/2532/")),
'full/97ee6f8a46cbbb418ea91502fd24176865cf39b2')
self.assertEqual(
file_path(Request("http://www.dorma.co.uk/images/product_details/2532")),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1')
self.assertEqual(
file_path(Request("http://www.dorma.co.uk/images/product_details/2532"),
response=Response("http://www.dorma.co.uk/images/product_details/2532"),
info=object()),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1')
self.assertEqual(
file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg.bohaha")),
'full/76c00cef2ef669ae65052661f68d451162829507')
self.assertEqual(
file_path(Request("data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAR0AAACxCAMAAADOHZloAAACClBMVEX/\
//+F0tzCwMK76ZKQ21AMqr7oAAC96JvD5aWM2kvZ78J0N7fmAAC46Y4Ap7y")),
'full/178059cbeba2e34120a67f2dc1afc3ecc09b61cb.png')
'full/178059cbeba2e34120a67f2dc1afc3ecc09b61cb.png')
def test_fs_store(self):
assert isinstance(self.pipeline.store, FSFilesStore)
@ -84,6 +93,7 @@ class FilesPipelineTestCase(unittest.TestCase):
result = yield self.pipeline.process_item(item, None)
self.assertEqual(result['files'][0]['checksum'], 'abc')
self.assertEqual(result['files'][0]['status'], 'uptodate')
for p in patchers:
p.stop()
@ -105,6 +115,29 @@ class FilesPipelineTestCase(unittest.TestCase):
result = yield self.pipeline.process_item(item, None)
self.assertNotEqual(result['files'][0]['checksum'], 'abc')
self.assertEqual(result['files'][0]['status'], 'downloaded')
for p in patchers:
p.stop()
@defer.inlineCallbacks
def test_file_cached(self):
item_url = "http://example.com/file3.pdf"
item = _create_item_with_files(item_url)
patchers = [
mock.patch.object(FilesPipeline, 'inc_stats', return_value=True),
mock.patch.object(FSFilesStore, 'stat_file', return_value={
'checksum': 'abc',
'last_modified': time.time() - (self.pipeline.expires * 60 * 60 * 24 * 2)}),
mock.patch.object(FilesPipeline, 'get_media_requests',
return_value=[_prepare_request_object(item_url, flags=['cached'])])
]
for p in patchers:
p.start()
result = yield self.pipeline.process_item(item, None)
self.assertNotEqual(result['files'][0]['checksum'], 'abc')
self.assertEqual(result['files'][0]['status'], 'cached')
for p in patchers:
p.stop()
@ -403,10 +436,10 @@ def _create_item_with_files(*files):
return item
def _prepare_request_object(item_url):
def _prepare_request_object(item_url, flags=None):
return Request(
item_url,
meta={'response': Response(item_url, status=200, body=b'data')})
meta={'response': Response(item_url, status=200, body=b'data', flags=flags)})
if __name__ == "__main__":

View File

@ -18,7 +18,7 @@ try:
except ImportError:
skip = 'Missing Python Imaging Library, install https://pypi.python.org/pypi/Pillow'
else:
encoders = set(('jpeg_encoder', 'jpeg_decoder'))
encoders = {'jpeg_encoder', 'jpeg_decoder'}
if not encoders.issubset(set(Image.core.__dict__)):
skip = 'Missing JPEG encoders'
@ -41,22 +41,29 @@ class ImagesPipelineTestCase(unittest.TestCase):
def test_file_path(self):
file_path = self.pipeline.file_path
self.assertEqual(file_path(Request("https://dev.mydeco.com/mydeco.gif")),
'full/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
self.assertEqual(file_path(Request("http://www.maddiebrown.co.uk///catalogue-items//image_54642_12175_95307.jpg")),
'full/0ffcd85d563bca45e2f90becd0ca737bc58a00b2.jpg')
self.assertEqual(file_path(Request("https://dev.mydeco.com/two/dirs/with%20spaces%2Bsigns.gif")),
'full/b250e3a74fff2e4703e310048a5b13eba79379d2.jpg')
self.assertEqual(file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg")),
'full/4507be485f38b0da8a0be9eb2e1dfab8a19223f2.jpg')
self.assertEqual(file_path(Request("http://www.dorma.co.uk/images/product_details/2532/")),
'full/97ee6f8a46cbbb418ea91502fd24176865cf39b2.jpg')
self.assertEqual(file_path(Request("http://www.dorma.co.uk/images/product_details/2532")),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1.jpg')
self.assertEqual(file_path(Request("http://www.dorma.co.uk/images/product_details/2532"),
response=Response("http://www.dorma.co.uk/images/product_details/2532"),
info=object()),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1.jpg')
self.assertEqual(
file_path(Request("https://dev.mydeco.com/mydeco.gif")),
'full/3fd165099d8e71b8a48b2683946e64dbfad8b52d.jpg')
self.assertEqual(
file_path(Request("http://www.maddiebrown.co.uk///catalogue-items//image_54642_12175_95307.jpg")),
'full/0ffcd85d563bca45e2f90becd0ca737bc58a00b2.jpg')
self.assertEqual(
file_path(Request("https://dev.mydeco.com/two/dirs/with%20spaces%2Bsigns.gif")),
'full/b250e3a74fff2e4703e310048a5b13eba79379d2.jpg')
self.assertEqual(
file_path(Request("http://www.dfsonline.co.uk/get_prod_image.php?img=status_0907_mdm.jpg")),
'full/4507be485f38b0da8a0be9eb2e1dfab8a19223f2.jpg')
self.assertEqual(
file_path(Request("http://www.dorma.co.uk/images/product_details/2532/")),
'full/97ee6f8a46cbbb418ea91502fd24176865cf39b2.jpg')
self.assertEqual(
file_path(Request("http://www.dorma.co.uk/images/product_details/2532")),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1.jpg')
self.assertEqual(
file_path(Request("http://www.dorma.co.uk/images/product_details/2532"),
response=Response("http://www.dorma.co.uk/images/product_details/2532"),
info=object()),
'full/244e0dd7d96a3b7b01f54eded250c9e272577aa1.jpg')
def test_thumbnail_name(self):
thumb_path = self.pipeline.thumb_path

View File

@ -4,6 +4,7 @@ import re
import sys
from subprocess import Popen, PIPE
from urllib.parse import urlsplit, urlunsplit
from unittest import skipIf
import pytest
from testfixtures import LogCapture
@ -56,6 +57,8 @@ def _wrong_credentials(proxy_url):
return urlunsplit(bad_auth_proxy)
@skipIf(sys.version_info < (3, 5, 4),
"requires mitmproxy < 3.0.0, which these tests do not support")
class ProxyConnectTestCase(TestCase):
def setUp(self):
@ -80,7 +83,7 @@ class ProxyConnectTestCase(TestCase):
yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
self._assert_got_response_code(200, log)
@pytest.mark.xfail(reason='Python 3.6+ fails this earlier', condition=sys.version_info.minor >= 6)
@pytest.mark.xfail(reason='Python 3.6+ fails this earlier', condition=sys.version_info >= (3, 6))
@defer.inlineCallbacks
def test_https_connect_tunnel_error(self):
crawler = get_crawler(SimpleSpider)

View File

@ -158,6 +158,12 @@ class CallbackKeywordArgumentsTestCase(TestCase):
if key in line.getMessage():
exceptions[key] = line
self.assertEqual(exceptions['takes_less'].exc_info[0], TypeError)
self.assertEqual(str(exceptions['takes_less'].exc_info[1]), "parse_takes_less() got an unexpected keyword argument 'number'")
self.assertEqual(
str(exceptions['takes_less'].exc_info[1]),
"parse_takes_less() got an unexpected keyword argument 'number'"
)
self.assertEqual(exceptions['takes_more'].exc_info[0], TypeError)
self.assertEqual(str(exceptions['takes_more'].exc_info[1]), "parse_takes_more() missing 1 required positional argument: 'other'")
self.assertEqual(
str(exceptions['takes_more'].exc_info[1]),
"parse_takes_more() missing 1 required positional argument: 'other'"
)

View File

@ -63,8 +63,9 @@ class ResponseTypesTest(unittest.TestCase):
def test_from_headers(self):
mappings = [
({'Content-Type': ['text/html; charset=utf-8']}, HtmlResponse),
({'Content-Type': ['application/octet-stream'], 'Content-Disposition': ['attachment; filename=data.txt']}, TextResponse),
({'Content-Type': ['text/html; charset=utf-8'], 'Content-Encoding': ['gzip']}, Response),
({'Content-Type': ['application/octet-stream'],
'Content-Disposition': ['attachment; filename=data.txt']}, TextResponse),
]
for source, cls in mappings:
source = Headers(source)
@ -76,8 +77,10 @@ class ResponseTypesTest(unittest.TestCase):
mappings = [
({'url': 'http://www.example.com/data.csv'}, TextResponse),
# headers takes precedence over url
({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), 'url': 'http://www.example.com/item/'}, HtmlResponse),
({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), 'url': 'http://www.example.com/page/'}, Response),
({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}),
'url': 'http://www.example.com/item/'}, HtmlResponse),
({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}),
'url': 'http://www.example.com/page/'}, Response),
]

View File

@ -19,18 +19,26 @@ class SelectorTestCase(unittest.TestCase):
for x in xl:
assert isinstance(x, Selector)
self.assertEqual(sel.xpath('//input').getall(),
[x.get() for x in sel.xpath('//input')])
self.assertEqual([x.get() for x in sel.xpath("//input[@name='a']/@name")],
[u'a'])
self.assertEqual([x.get() for x in sel.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
[u'12.0'])
self.assertEqual(sel.xpath("concat('xpath', 'rules')").getall(),
[u'xpathrules'])
self.assertEqual([x.get() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
[u'12'])
self.assertEqual(
sel.xpath('//input').getall(),
[x.get() for x in sel.xpath('//input')]
)
self.assertEqual(
[x.get() for x in sel.xpath("//input[@name='a']/@name")],
[u'a']
)
self.assertEqual(
[x.get() for x in sel.xpath("number(concat(//input[@name='a']/@value, //input[@name='b']/@value))")],
[u'12.0']
)
self.assertEqual(
sel.xpath("concat('xpath', 'rules')").getall(),
[u'xpathrules']
)
self.assertEqual(
[x.get() for x in sel.xpath("concat(//input[@name='a']/@value, //input[@name='b']/@value)")],
[u'12']
)
def test_root_base_url(self):
body = b'<html><form action="/path"><input name="a" /></form></html>'

View File

@ -120,7 +120,9 @@ class XMLFeedSpiderTest(SpiderTest):
body = b"""<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns:x="http://www.google.com/schemas/sitemap/0.84"
xmlns:y="http://www.example.com/schemas/extras/1.0">
<url><x:loc>http://www.example.com/Special-Offers.html</loc><y:updated>2009-08-16</updated><other value="bar" y:custom="fuu"/></url>
<url><x:loc>http://www.example.com/Special-Offers.html</loc><y:updated>2009-08-16</updated>
<other value="bar" y:custom="fuu"/>
</url>
<url><loc>http://www.example.com/</loc><y:updated>2009-08-16</updated><other value="foo"/></url>
</urlset>"""
response = XmlResponse(url='http://example.com/sitemap.xml', body=body)

View File

@ -42,7 +42,7 @@ class SpiderLoaderTest(unittest.TestCase):
def test_list(self):
self.assertEqual(
set(self.spider_loader.list()),
set(['spider1', 'spider2', 'spider3', 'spider4']))
{'spider1', 'spider2', 'spider3', 'spider4'})
def test_load(self):
spider1 = self.spider_loader.load("spider1")
@ -57,7 +57,7 @@ class SpiderLoaderTest(unittest.TestCase):
['spider2'])
self.assertEqual(
set(self.spider_loader.find_by_request(Request('http://scrapy3.org/test'))),
set(['spider1', 'spider2']))
{'spider1', 'spider2'})
self.assertEqual(
self.spider_loader.find_by_request(Request('http://scrapy999.org/test')),
[])
@ -151,7 +151,7 @@ class DuplicateSpiderNameLoaderTest(unittest.TestCase):
self.assertNotIn("'spider4'", msg)
spiders = set(spider_loader.list())
self.assertEqual(spiders, set(['spider1', 'spider2', 'spider3', 'spider4']))
self.assertEqual(spiders, {'spider1', 'spider2', 'spider3', 'spider4'})
def test_multiple_dupename_warning(self):
# copy 2 spider modules so as to have duplicate spider name
@ -177,4 +177,4 @@ class DuplicateSpiderNameLoaderTest(unittest.TestCase):
self.assertNotIn("'spider4'", msg)
spiders = set(spider_loader.list())
self.assertEqual(spiders, set(['spider1', 'spider2', 'spider3', 'spider4']))
self.assertEqual(spiders, {'spider1', 'spider2', 'spider3', 'spider4'})

View File

@ -385,9 +385,15 @@ class TestSpiderMiddleware(TestCase):
log4 = yield self.crawl_log(GeneratorOutputChainSpider)
self.assertIn("'item_scraped_count': 2", str(log4))
self.assertIn("GeneratorRecoverMiddleware.process_spider_exception: LookupError caught", str(log4))
self.assertIn("GeneratorDoNothingAfterFailureMiddleware.process_spider_exception: LookupError caught", str(log4))
self.assertNotIn("GeneratorFailMiddleware.process_spider_exception: LookupError caught", str(log4))
self.assertNotIn("GeneratorDoNothingAfterRecoveryMiddleware.process_spider_exception: LookupError caught", str(log4))
self.assertIn(
"GeneratorDoNothingAfterFailureMiddleware.process_spider_exception: LookupError caught",
str(log4))
self.assertNotIn(
"GeneratorFailMiddleware.process_spider_exception: LookupError caught",
str(log4))
self.assertNotIn(
"GeneratorDoNothingAfterRecoveryMiddleware.process_spider_exception: LookupError caught",
str(log4))
item_from_callback = {'processed': [
'parse-first-item',
'GeneratorFailMiddleware.process_spider_output',
@ -414,9 +420,13 @@ class TestSpiderMiddleware(TestCase):
log5 = yield self.crawl_log(NotGeneratorOutputChainSpider)
self.assertIn("'item_scraped_count': 1", str(log5))
self.assertIn("GeneratorRecoverMiddleware.process_spider_exception: ReferenceError caught", str(log5))
self.assertIn("GeneratorDoNothingAfterFailureMiddleware.process_spider_exception: ReferenceError caught", str(log5))
self.assertIn(
"GeneratorDoNothingAfterFailureMiddleware.process_spider_exception: ReferenceError caught",
str(log5))
self.assertNotIn("GeneratorFailMiddleware.process_spider_exception: ReferenceError caught", str(log5))
self.assertNotIn("GeneratorDoNothingAfterRecoveryMiddleware.process_spider_exception: ReferenceError caught", str(log5))
self.assertNotIn(
"GeneratorDoNothingAfterRecoveryMiddleware.process_spider_exception: ReferenceError caught",
str(log5))
item_recovered = {'processed': [
'NotGeneratorRecoverMiddleware.process_spider_exception',
'NotGeneratorDoNothingAfterRecoveryMiddleware.process_spider_output']}

View File

@ -119,7 +119,11 @@ class MixinSameOrigin:
('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'),
('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'),
('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'),
('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'),
(
'http://example.com:8888/page.html',
'http://example.com:8888/not-page.html',
b'http://example.com:8888/page.html',
),
# Different host: do NOT send referrer
('https://example.com/page.html', 'https://not.example.com/otherpage.html', None),
@ -139,8 +143,12 @@ class MixinSameOrigin:
('ftps://example.com/urls.zip', 'https://example.com/not-page.html', None),
# test for user/password stripping
('https://user:password@example.com/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'),
('https://user:password@example.com/page.html', 'http://example.com/not-page.html', None),
(
'https://user:password@example.com/page.html',
'https://example.com/not-page.html',
b'https://example.com/page.html',
),
]
@ -184,7 +192,11 @@ class MixinOriginWhenCrossOrigin:
('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'),
('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'),
('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'),
('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'),
(
'http://example.com:8888/page.html',
'http://example.com:8888/not-page.html',
b'http://example.com:8888/page.html',
),
# Different host: send origin as referrer
('https://example2.com/page.html', 'https://scrapy.org/otherpage.html', b'https://example2.com/'),
@ -205,9 +217,17 @@ class MixinOriginWhenCrossOrigin:
('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'),
# test for user/password stripping
('https://user:password@example5.com/page.html', 'https://example5.com/not-page.html', b'https://example5.com/page.html'),
(
'https://user:password@example5.com/page.html',
'https://example5.com/not-page.html',
b'https://example5.com/page.html',
),
# TLS to non-TLS downgrade: send origin
('https://user:password@example5.com/page.html', 'http://example5.com/not-page.html', b'https://example5.com/'),
(
'https://user:password@example5.com/page.html',
'http://example5.com/not-page.html',
b'https://example5.com/',
),
]
@ -219,7 +239,11 @@ class MixinStrictOriginWhenCrossOrigin:
('https://example.com:443/page.html', 'https://example.com/not-page.html', b'https://example.com/page.html'),
('http://example.com:80/page.html', 'http://example.com/not-page.html', b'http://example.com/page.html'),
('http://example.com/page.html', 'http://example.com:80/not-page.html', b'http://example.com/page.html'),
('http://example.com:8888/page.html', 'http://example.com:8888/not-page.html', b'http://example.com:8888/page.html'),
(
'http://example.com:8888/page.html',
'http://example.com:8888/not-page.html',
b'http://example.com:8888/page.html',
),
# Different host: send origin as referrer
('https://example2.com/page.html', 'https://scrapy.org/otherpage.html', b'https://example2.com/'),
@ -248,7 +272,11 @@ class MixinStrictOriginWhenCrossOrigin:
('ftps://example4.com/urls.zip', 'https://example4.com/not-page.html', b'ftps://example4.com/'),
# test for user/password stripping
('https://user:password@example5.com/page.html', 'https://example5.com/not-page.html', b'https://example5.com/page.html'),
(
'https://user:password@example5.com/page.html',
'https://example5.com/not-page.html',
b'https://example5.com/page.html',
),
# TLS to non-TLS downgrade: send nothing
('https://user:password@example5.com/page.html', 'http://example5.com/not-page.html', None),
@ -281,8 +309,16 @@ class MixinUnsafeUrl:
('ftp://example3.com/urls.zip', 'https://scrapy.org/', b'ftp://example3.com/urls.zip'),
# test for user/password stripping
('http://user:password@example4.com/page.html', 'https://not.example4.com/', b'http://example4.com/page.html'),
('https://user:password@example4.com/page.html', 'http://scrapy.org/', b'https://example4.com/page.html'),
(
'http://user:password@example4.com/page.html',
'https://not.example4.com/',
b'http://example4.com/page.html',
),
(
'https://user:password@example4.com/page.html',
'http://scrapy.org/',
b'https://example4.com/page.html',
),
]

View File

@ -217,7 +217,7 @@ class SequenceExcludeTest(unittest.TestCase):
def test_set(self):
"""Anything that is not in the supplied sequence will evaluate as 'in' the container."""
seq = set([-3, "test", 1.1])
seq = {-3, "test", 1.1}
d = SequenceExclude(seq)
self.assertIn(0, d)
self.assertIn("foo", d)

View File

@ -16,7 +16,8 @@ class XmliterTestCase(unittest.TestCase):
def test_xmliter(self):
body = b"""<?xml version="1.0" encoding="UTF-8"?>\
<products xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xsi:noNamespaceSchemaLocation="someschmea.xsd">\
<products xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
xsi:noNamespaceSchemaLocation="someschmea.xsd">\
<product id="001">\
<type>Type 1</type>\
<name>Name 1</name>\
@ -105,7 +106,10 @@ class XmliterTestCase(unittest.TestCase):
(u'27', [u'A'], [u'27'])])
def test_xmliter_text(self):
body = u"""<?xml version="1.0" encoding="UTF-8"?><products><product>one</product><product>two</product></products>"""
body = (
'<?xml version="1.0" encoding="UTF-8"?>'
'<products><product>one</product><product>two</product></products>'
)
self.assertEqual([x.xpath("text()").getall() for x in self.xmliter(body, 'product')],
[[u'one'], [u'two']])
@ -137,7 +141,10 @@ class XmliterTestCase(unittest.TestCase):
self.assertEqual(node.xpath('title/text()').getall(), ['Item 1'])
self.assertEqual(node.xpath('description/text()').getall(), ['This is item 1'])
self.assertEqual(node.xpath('link/text()').getall(), ['http://www.mydummycompany.com/items/1'])
self.assertEqual(node.xpath('g:image_link/text()').getall(), ['http://www.mydummycompany.com/images/item1.jpg'])
self.assertEqual(
node.xpath('g:image_link/text()').getall(),
['http://www.mydummycompany.com/images/item1.jpg']
)
self.assertEqual(node.xpath('g:id/text()').getall(), ['ITEM_1'])
self.assertEqual(node.xpath('g:price/text()').getall(), ['400'])
self.assertEqual(node.xpath('image_link/text()').getall(), [])
@ -145,7 +152,10 @@ class XmliterTestCase(unittest.TestCase):
self.assertEqual(node.xpath('price/text()').getall(), [])
def test_xmliter_exception(self):
body = u"""<?xml version="1.0" encoding="UTF-8"?><products><product>one</product><product>two</product></products>"""
body = (
'<?xml version="1.0" encoding="UTF-8"?>'
'<products><product>one</product><product>two</product></products>'
)
iter = self.xmliter(body, 'product')
next(iter)
@ -158,7 +168,12 @@ class XmliterTestCase(unittest.TestCase):
self.assertRaises(TypeError, next, i)
def test_xmliter_encoding(self):
body = b'<?xml version="1.0" encoding="ISO-8859-9"?>\n<xml>\n <item>Some Turkish Characters \xd6\xc7\xde\xdd\xd0\xdc \xfc\xf0\xfd\xfe\xe7\xf6</item>\n</xml>\n\n'
body = (
b'<?xml version="1.0" encoding="ISO-8859-9"?>\n'
b'<xml>\n'
b' <item>Some Turkish Characters \xd6\xc7\xde\xdd\xd0\xdc \xfc\xf0\xfd\xfe\xe7\xf6</item>\n'
b'</xml>\n\n'
)
response = XmlResponse('http://www.example.com', body=body)
self.assertEqual(
next(self.xmliter(response, 'item')).get(),

View File

@ -26,20 +26,20 @@ class UtilsMiscTestCase(unittest.TestCase):
'tests.test_utils_misc.test_walk_modules.mod.mod0',
'tests.test_utils_misc.test_walk_modules.mod1',
]
self.assertEqual(set([m.__name__ for m in mods]), set(expected))
self.assertEqual({m.__name__ for m in mods}, set(expected))
mods = walk_modules('tests.test_utils_misc.test_walk_modules.mod')
expected = [
'tests.test_utils_misc.test_walk_modules.mod',
'tests.test_utils_misc.test_walk_modules.mod.mod0',
]
self.assertEqual(set([m.__name__ for m in mods]), set(expected))
self.assertEqual({m.__name__ for m in mods}, set(expected))
mods = walk_modules('tests.test_utils_misc.test_walk_modules.mod1')
expected = [
'tests.test_utils_misc.test_walk_modules.mod1',
]
self.assertEqual(set([m.__name__ for m in mods]), set(expected))
self.assertEqual({m.__name__ for m in mods}, set(expected))
self.assertRaises(ImportError, walk_modules, 'nomodule999')
@ -54,7 +54,7 @@ class UtilsMiscTestCase(unittest.TestCase):
'testegg.spiders.b',
'testegg'
]
self.assertEqual(set([m.__name__ for m in mods]), set(expected))
self.assertEqual({m.__name__ for m in mods}, set(expected))
finally:
sys.path.remove(egg)

View File

@ -76,8 +76,12 @@ class UtilsRequestTest(unittest.TestCase):
r1 = Request("http://www.example.com/some/page.html?arg=1")
self.assertEqual(request_httprepr(r1), b'GET /some/page.html?arg=1 HTTP/1.1\r\nHost: www.example.com\r\n\r\n')
r1 = Request("http://www.example.com", method='POST', headers={"Content-type": b"text/html"}, body=b"Some body")
self.assertEqual(request_httprepr(r1), b'POST / HTTP/1.1\r\nHost: www.example.com\r\nContent-Type: text/html\r\n\r\nSome body')
r1 = Request("http://www.example.com", method='POST',
headers={"Content-type": b"text/html"}, body=b"Some body")
self.assertEqual(
request_httprepr(r1),
b'POST / HTTP/1.1\r\nHost: www.example.com\r\nContent-Type: text/html\r\n\r\nSome body'
)
def test_request_httprepr_for_non_http_request(self):
# the representation is not important but it must not fail.

View File

@ -26,7 +26,8 @@ class SitemapTest(unittest.TestCase):
list(s),
[
{'priority': '1', 'loc': 'http://www.example.com/', 'lastmod': '2009-08-16', 'changefreq': 'daily'},
{'priority': '0.8', 'loc': 'http://www.example.com/Special-Offers.html', 'lastmod': '2009-08-16', 'changefreq': 'weekly'},
{'priority': '0.8', 'loc': 'http://www.example.com/Special-Offers.html',
'lastmod': '2009-08-16', 'changefreq': 'weekly'},
]
)
@ -43,7 +44,13 @@ class SitemapTest(unittest.TestCase):
</sitemap>
</sitemapindex>""")
assert s.type == 'sitemapindex'
self.assertEqual(list(s), [{'loc': 'http://www.example.com/sitemap1.xml.gz', 'lastmod': '2004-10-01T18:23:17+00:00'}, {'loc': 'http://www.example.com/sitemap2.xml.gz', 'lastmod': '2005-01-01'}])
self.assertEqual(
list(s),
[
{'loc': 'http://www.example.com/sitemap1.xml.gz', 'lastmod': '2004-10-01T18:23:17+00:00'},
{'loc': 'http://www.example.com/sitemap2.xml.gz', 'lastmod': '2005-01-01'},
]
)
def test_sitemap_strip(self):
"""Assert we can deal with trailing spaces inside <loc> tags - we've

View File

@ -2,7 +2,7 @@ import unittest
from scrapy import Spider
from scrapy.http import Request
from scrapy.item import BaseItem
from scrapy.item import Item
from scrapy.utils.spider import iterate_spider_output, iter_spider_classes
@ -17,7 +17,7 @@ class MySpider2(Spider):
class UtilsSpidersTestCase(unittest.TestCase):
def test_iterate_spider_output(self):
i = BaseItem()
i = Item()
r = Request('http://scrapytest.org')
o = object()

View File

@ -27,7 +27,10 @@ class UrlUtilsTest(unittest.TestCase):
self.assertTrue(url_is_from_any_domain(url, ['192.169.0.15:8080']))
self.assertFalse(url_is_from_any_domain(url, ['192.169.0.15']))
url = 'javascript:%20document.orderform_2581_1190810811.mode.value=%27add%27;%20javascript:%20document.orderform_2581_1190810811.submit%28%29'
url = (
'javascript:%20document.orderform_2581_1190810811.mode.value=%27add%27;%20'
'javascript:%20document.orderform_2581_1190810811.submit%28%29'
)
self.assertFalse(url_is_from_any_domain(url, ['testdomain.com']))
self.assertFalse(url_is_from_any_domain(url + '.testdomain.com', ['testdomain.com']))
@ -55,7 +58,7 @@ class UrlUtilsTest(unittest.TestCase):
self.assertTrue(url_is_from_spider('http://www.example.net/some/page.html', spider))
self.assertFalse(url_is_from_spider('http://www.example.us/some/page.html', spider))
spider = Spider(name='example.com', allowed_domains=set(('example.com', 'example.net')))
spider = Spider(name='example.com', allowed_domains={'example.com', 'example.net'})
self.assertTrue(url_is_from_spider('http://www.example.com/some/page.html', spider))
spider = Spider(name='example.com', allowed_domains=('example.com', 'example.net'))