diff --git a/.travis.yml b/.travis.yml index e857abbd8..ae9c745ac 100644 --- a/.travis.yml +++ b/.travis.yml @@ -1,5 +1,5 @@ language: python -python: 2.7 +python: 3.5 sudo: false branches: only: @@ -9,6 +9,7 @@ env: - TOXENV=py27 - TOXENV=precise - TOXENV=py33 + - TOXENV=py35 - TOXENV=docs install: - pip install -U tox twine wheel codecov diff --git a/conftest.py b/conftest.py index f9ca3ab93..8b4faf8fc 100644 --- a/conftest.py +++ b/conftest.py @@ -1,6 +1,7 @@ import glob import six import pytest +from twisted import version as twisted_version def _py_files(folder): @@ -26,11 +27,14 @@ collect_ignore = [ ] + _py_files("scrapy/contrib") + _py_files("scrapy/contrib_exp") +if (twisted_version.major, twisted_version.minor, twisted_version.micro) >= (15, 5, 0): + collect_ignore += _py_files("scrapy/xlib/tx") + if six.PY3: for line in open('tests/py3-ignores.txt'): file_path = line.strip() - if len(file_path) > 0 and file_path[0] != '#': + if file_path and file_path[0] != '#': collect_ignore.append(file_path) diff --git a/docs/faq.rst b/docs/faq.rst index 2e61f44ee..3d2bd8d4d 100644 --- a/docs/faq.rst +++ b/docs/faq.rst @@ -144,7 +144,7 @@ I get "Filtered offsite request" messages. How can I fix them? Those messages (logged with ``DEBUG`` level) don't necessarily mean there is a problem, so you may not need to fix them. -Those message are thrown by the Offsite Spider Middleware, which is a spider +Those messages are thrown by the Offsite Spider Middleware, which is a spider middleware (enabled by default) whose purpose is to filter out requests to domains outside the ones covered by the spider. diff --git a/docs/index.rst b/docs/index.rst index 0d21f5d40..4cb3eb741 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -28,6 +28,7 @@ First steps =========== .. toctree:: + :caption: First steps :hidden: intro/overview @@ -53,6 +54,7 @@ Basic concepts ============== .. toctree:: + :caption: Basic concepts :hidden: topics/commands @@ -110,6 +112,7 @@ Built-in services ================= .. toctree:: + :caption: Built-in services :hidden: topics/logging @@ -138,6 +141,7 @@ Solving specific problems ========================= .. toctree:: + :caption: Solving specific problems :hidden: faq @@ -203,6 +207,7 @@ Extending Scrapy ================ .. toctree:: + :caption: Extending Scrapy :hidden: topics/architecture @@ -240,6 +245,7 @@ All the rest ============ .. toctree:: + :caption: All the rest :hidden: news diff --git a/docs/intro/install.rst b/docs/intro/install.rst index 3adb4e6b0..122de47f6 100644 --- a/docs/intro/install.rst +++ b/docs/intro/install.rst @@ -24,9 +24,7 @@ The installation steps assume that you have the following things installed: where the Python installer ships it bundled. You can install Scrapy using pip (which is the canonical way to install Python -packages). - -To install using pip:: +packages). To install using ``pip`` run:: pip install Scrapy @@ -35,6 +33,22 @@ To install using pip:: Platform specific installation notes ==================================== +Anaconda +-------- + +.. note:: + + For Windows users, or if you have issues installing through `pip`, this is + the recommended way to install Scrapy. + +If you already have installed `Anaconda`_ or `Miniconda`_, the company +`Scrapinghub`_ maintains official conda packages for Linux, Windows and OS X. + +To install Scrapy using ``conda``, run:: + + conda install -c scrapinghub scrapy + + Windows ------- @@ -167,3 +181,6 @@ After any of these workarounds you should be able to install Scrapy:: .. _homebrew: http://brew.sh/ .. _zsh: http://www.zsh.org/ .. _virtualenv: https://virtualenv.pypa.io/en/latest/ +.. _Scrapinghub: http://scrapinghub.com +.. _Anaconda: http://docs.continuum.io/anaconda/index +.. _Miniconda: http://conda.pydata.org/docs/install/quick.html diff --git a/docs/news.rst b/docs/news.rst index 5df1b1a6a..4d7dc4d41 100644 --- a/docs/news.rst +++ b/docs/news.rst @@ -3,6 +3,58 @@ Release notes ============= +1.0.4 (2015-12-30) +------------------ + +- Ignoring xlib/tx folder, depending on Twisted version. (:commit:`7dfa979`) +- Run on new travis-ci infra (:commit:`6e42f0b`) +- Spelling fixes (:commit:`823a1cc`) +- escape nodename in xmliter regex (:commit:`da3c155`) +- test xml nodename with dots (:commit:`4418fc3`) +- TST don't use broken Pillow version in tests (:commit:`a55078c`) +- disable log on version command. closes #1426 (:commit:`86fc330`) +- disable log on startproject command (:commit:`db4c9fe`) +- Add PyPI download stats badge (:commit:`df2b944`) +- don't run tests twice on Travis if a PR is made from a scrapy/scrapy branch (:commit:`a83ab41`) +- Add Python 3 porting status badge to the README (:commit:`73ac80d`) +- fixed RFPDupeFilter persistence (:commit:`97d080e`) +- TST a test to show that dupefilter persistence is not working (:commit:`97f2fb3`) +- explicit close file on file:// scheme handler (:commit:`d9b4850`) +- Disable dupefilter in shell (:commit:`c0d0734`) +- DOC: Add captions to toctrees which appear in sidebar (:commit:`aa239ad`) +- DOC Removed pywin32 from install instructions as it's already declared as dependency. (:commit:`10eb400`) +- Added installation notes about using Conda for Windows and other OSes. (:commit:`1c3600a`) +- Fixed minor grammar issues. (:commit:`7f4ddd5`) +- fixed a typo in the documentation. (:commit:`b71f677`) +- Version 1 now exists (:commit:`5456c0e`) +- fix another invalid xpath error (:commit:`0a1366e`) +- fix ValueError: Invalid XPath: //div/[id="not-exists"]/text() on selectors.rst (:commit:`ca8d60f`) +- Typos corrections (:commit:`7067117`) +- fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`) +- Add note to ubuntu install section about debian compatibility (:commit:`23fda69`) +- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`) +- Reference Homebrew's homepage for installation instructions (:commit:`1925db1`) +- Add oldest supported tox version to contributing docs (:commit:`5d10d6d`) +- Note in install docs about pip being already included in python>=2.7.9 (:commit:`85c980e`) +- Add non-python dependencies to Ubuntu install section in the docs (:commit:`fbd010d`) +- Add OS X installation section to docs (:commit:`d8f4cba`) +- DOC(ENH): specify path to rtd theme explicitly (:commit:`de73b1a`) +- minor: scrapy.Spider docs grammar (:commit:`1ddcc7b`) +- Make common practices sample code match the comments (:commit:`1b85bcf`) +- nextcall repetitive calls (heartbeats). (:commit:`55f7104`) +- Backport fix compatibility with Twisted 15.4.0 (:commit:`b262411`) +- pin pytest to 2.7.3 (:commit:`a6535c2`) +- Merge pull request #1512 from mgedmin/patch-1 (:commit:`8876111`) +- Merge pull request #1513 from mgedmin/patch-2 (:commit:`5d4daf8`) +- Typo (:commit:`f8d0682`) +- Fix list formatting (:commit:`5f83a93`) +- fix scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`) +- Merge pull request #1475 from rweindl/patch-1 (:commit:`2d688cd`) +- Update tutorial.rst (:commit:`fbc1f25`) +- Merge pull request #1449 from rhoekman/patch-1 (:commit:`7d6538c`) +- Small grammatical change (:commit:`8752294`) +- Add openssl version to version command (:commit:`13c45ac`) + 1.0.3 (2015-08-11) ------------------ diff --git a/docs/topics/broad-crawls.rst b/docs/topics/broad-crawls.rst index aaf46bc92..28ed7c064 100644 --- a/docs/topics/broad-crawls.rst +++ b/docs/topics/broad-crawls.rst @@ -34,7 +34,7 @@ These are some common properties often found in broad crawls: As said above, Scrapy default settings are optimized for focused crawls, not broad crawls. However, due to its asynchronous architecture, Scrapy is very -well suited for performing fast broad crawls. This page summarize some things +well suited for performing fast broad crawls. This page summarizes some things you need to keep in mind when using Scrapy for doing broad crawls, along with concrete suggestions of Scrapy settings to tune in order to achieve an efficient broad crawl. @@ -46,11 +46,11 @@ Concurrency is the number of requests that are processed in parallel. There is a global limit and a per-domain limit. The default global concurrency limit in Scrapy is not suitable for crawling -many different domains in parallel, so you will want to increase it. How much +many different domains in parallel, so you will want to increase it. How much to increase it will depend on how much CPU you crawler will have available. A good starting point is ``100``, but the best way to find out is by doing some trials and identifying at what concurrency your Scrapy process gets CPU -bounded. For optimum performance, You should pick a concurrency where CPU usage +bounded. For optimum performance, you should pick a concurrency where CPU usage is at 80-90%. To increase the global concurrency use:: diff --git a/docs/topics/downloader-middleware.rst b/docs/topics/downloader-middleware.rst index 08d8f3edf..a97d5a696 100644 --- a/docs/topics/downloader-middleware.rst +++ b/docs/topics/downloader-middleware.rst @@ -23,20 +23,22 @@ Here's an example:: 'myproject.middlewares.CustomDownloaderMiddleware': 543, } -The specified :setting:`DOWNLOADER_MIDDLEWARES` setting is merged with the -default one (i.e. it does not overwrite it) and then sorted by order to get the -final sorted list of enabled middlewares: the first middleware is the one -closer to the engine and the last is the one closer to the downloader. +The :setting:`DOWNLOADER_MIDDLEWARES` setting is merged with the +:setting:`DOWNLOADER_MIDDLEWARES_BASE` setting defined in Scrapy (and not meant +to be overridden) and then sorted by order to get the final sorted list of +enabled middlewares: the first middleware is the one closer to the engine and +the last is the one closer to the downloader. -To decide which order to assign to your middleware see the default -:setting:`DOWNLOADER_MIDDLEWARES` setting and pick a value according to +To decide which order to assign to your middleware see the +:setting:`DOWNLOADER_MIDDLEWARES_BASE` setting and pick a value according to where you want to insert the middleware. The order does matter because each middleware performs a different action and your middleware could depend on some previous (or subsequent) middleware being applied. -If you want to disable a built-in middleware you must define it in your -project's :setting:`DOWNLOADER_MIDDLEWARES` setting and assign ``None`` as its -value. For example, if you want to disable the user-agent middleware:: +If you want to disable a built-in middleware (the ones defined in +:setting:`DOWNLOADER_MIDDLEWARES_BASE` and enabled by default) you must define it +in your project's :setting:`DOWNLOADER_MIDDLEWARES` setting and assign `None` +as its value. For example, if you want to disable the user-agent middleware:: DOWNLOADER_MIDDLEWARES = { 'myproject.middlewares.CustomDownloaderMiddleware': 543, @@ -162,7 +164,7 @@ middleware, see the :ref:`downloader middleware usage guide `. For a list of the components enabled by default (and their orders) see the -:setting:`DOWNLOADER_MIDDLEWARES` setting. +:setting:`DOWNLOADER_MIDDLEWARES_BASE` setting. .. _cookies-mw: @@ -785,14 +787,16 @@ Default: ``True`` Whether the Meta Refresh middleware will be enabled. -.. setting:: REDIRECT_MAX_METAREFRESH_DELAY +.. setting:: METAREFRESH_MAXDELAY -REDIRECT_MAX_METAREFRESH_DELAY -^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^ +METAREFRESH_MAXDELAY +^^^^^^^^^^^^^^^^^^^^ Default: ``100`` The maximum meta-refresh delay (in seconds) to follow the redirection. +Some sites use meta-refresh for redirecting to a session expired page, so we +restrict automatic redirection to the maximum delay. RetryMiddleware --------------- @@ -947,6 +951,18 @@ Default: ``False`` Whether the AjaxCrawlMiddleware will be enabled. You may want to enable it for :ref:`broad crawls `. +HttpProxyMiddleware settings +~~~~~~~~~~~~~~~~~~~~~~~~~~~~ + +.. setting:: HTTPPROXY_AUTH_ENCODING + +HTTPPROXY_AUTH_ENCODING +^^^^^^^^^^^^^^^^^^^^^^^ + +Default: ``"latin-1"`` + +The default encoding for proxy authentication on :class:`HttpProxyMiddleware`. + .. _DBM: http://en.wikipedia.org/wiki/Dbm .. _anydbm: https://docs.python.org/2/library/anydbm.html diff --git a/docs/topics/extensions.rst b/docs/topics/extensions.rst index 11c0aadb6..14528ba9a 100644 --- a/docs/topics/extensions.rst +++ b/docs/topics/extensions.rst @@ -17,7 +17,7 @@ Extensions use the :ref:`Scrapy settings ` to manage their settings, just like any other Scrapy code. It is customary for extensions to prefix their settings with their own name, to -avoid collision with existing (and future) extensions. For example, an +avoid collision with existing (and future) extensions. For example, a hypothetic extension to handle `Google Sitemaps`_ would use settings like `GOOGLESITEMAP_ENABLED`, `GOOGLESITEMAP_DEPTH`, and so on. @@ -42,13 +42,14 @@ by a string: the full Python path to the extension's class name. For example:: As you can see, the :setting:`EXTENSIONS` setting is a dict where the keys are the extension paths, and their values are the orders, which define the -extension *loading* order. The specified :setting:`EXTENSIONS` setting is merged -with the default one (i.e. it does not overwrite it) and then sorted by order -to get the final sorted list of enabled extensions. +extension *loading* order. The :setting:`EXTENSIONS` setting is merged with the +:setting:`EXTENSIONS_BASE` setting defined in Scrapy (and not meant to be +overridden) and then sorted by order to get the final sorted list of enabled +extensions. As extensions typically do not depend on each other, their loading order is -irrelevant in most cases. This is why the default :setting:`EXTENSIONS` setting -defines all extensions with the same order (``500``). However, this feature can +irrelevant in most cases. This is why the :setting:`EXTENSIONS_BASE` setting +defines all extensions with the same order (``0``). However, this feature can be exploited if you need to add an extension which depends on other extensions already loaded. @@ -63,7 +64,7 @@ Disabling an extension ====================== In order to disable an extension that comes enabled by default (ie. those -included in the default :setting:`EXTENSIONS` setting) you must set its order to +included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to ``None``. For example:: EXTENSIONS = { @@ -143,7 +144,7 @@ Here is the code of such extension:: self.items_scraped += 1 if self.items_scraped % self.item_count == 0: logger.info("scraped %d items", self.items_scraped) - + .. _topics-extensions-ref: diff --git a/docs/topics/feed-exports.rst b/docs/topics/feed-exports.rst index d8b8da166..03c6fb3fb 100644 --- a/docs/topics/feed-exports.rst +++ b/docs/topics/feed-exports.rst @@ -265,6 +265,16 @@ Whether to export empty feeds (ie. feeds with no items). FEED_STORAGES ------------- +Default:: ``{}`` + +A dict containing additional feed storage backends supported by your project. +The keys are URI schemes and the values are paths to storage classes. + +.. setting:: FEED_STORAGES_BASE + +FEED_STORAGES_BASE +------------------ + Default:: { @@ -275,19 +285,30 @@ Default:: 'ftp': 'scrapy.extensions.feedexport.FTPFeedStorage', } -A dict containing all feed storage backends supported by your project. The keys -are URI schemes and the values are paths to storage classes. +A dict containing the built-in feed storage backends supported by Scrapy. You +can disable any of these backends by assigning ``None`` to their URI scheme in +:setting:`FEED_STORAGES`. E.g., to disable the built-in FTP storage backend +(without replacement), place this in your ``settings.py``:: -When you set :setting:`FEED_STORAGES` manually, e.g. in your project's settings -module, it will be merged with the default, not overwrite it. If you want to -disable any of the default feed storage backends, you must assign ``None`` as -their value. + FEED_STORAGES = { + 'ftp': None, + } .. setting:: FEED_EXPORTERS FEED_EXPORTERS -------------- +Default:: ``{}`` + +A dict containing additional exporters supported by your project. The keys are +serialization formats and the values are paths to :ref:`Item exporter +` classes. + +.. setting:: FEED_EXPORTERS_BASE + +FEED_EXPORTERS_BASE +------------------- Default:: { @@ -300,14 +321,14 @@ Default:: 'pickle': 'scrapy.exporters.PickleItemExporter', } -A dict containing all feed exporters supported by your project. The keys are -URI schemes and the values are paths to :ref:`Item exporter ` -classes. +A dict containing the built-in feed exporters supported by Scrapy. You can +disable any of these exporters by assigning ``None`` to their serialization +format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter +(without replacement), place this in your ``settings.py``:: -When you set :setting:`FEED_EXPORTERS` manually, e.g. in your project's settings -module, it will be merged with the default, not overwrite it. If you want to -disable any of the default feed exporters, you must assign ``None`` as their -value. + FEED_EXPORTERS = { + 'csv': None, + } .. _URI: http://en.wikipedia.org/wiki/Uniform_Resource_Identifier .. _Amazon S3: http://aws.amazon.com/s3/ diff --git a/docs/topics/item-pipeline.rst b/docs/topics/item-pipeline.rst index f74400b4d..28969be61 100644 --- a/docs/topics/item-pipeline.rst +++ b/docs/topics/item-pipeline.rst @@ -95,7 +95,7 @@ contain a price:: Write items to a JSON file -------------------------- -The following pipeline stores all scraped items (from all spiders) into a a +The following pipeline stores all scraped items (from all spiders) into a single ``items.jl`` file, containing one item per line serialized in JSON format:: diff --git a/docs/topics/items.rst b/docs/topics/items.rst index 21ec0ed8c..4a8f47e93 100644 --- a/docs/topics/items.rst +++ b/docs/topics/items.rst @@ -61,7 +61,7 @@ the example above. You can specify any kind of metadata for each field. There is no restriction on the values accepted by :class:`Field` objects. For this same reason, there is no reference list of all available metadata keys. Each key -defined in :class:`Field` objects could be used by a different components, and +defined in :class:`Field` objects could be used by a different component, and only those components know about it. You can also define and use any other :class:`Field` key in your project too, for your own needs. The main goal of :class:`Field` objects is to provide a way to define all field metadata in one diff --git a/docs/topics/leaks.rst b/docs/topics/leaks.rst index 735137ea2..92590c180 100644 --- a/docs/topics/leaks.rst +++ b/docs/topics/leaks.rst @@ -97,7 +97,7 @@ subclasses): A real example -------------- -Let's see a concrete example of an hypothetical case of memory leaks. +Let's see a concrete example of a hypothetical case of memory leaks. Suppose we have some spider with a line similar to this one:: return Request("http://www.somenastyspider.com/product.php?pid=%d" % product_id, diff --git a/docs/topics/practices.rst b/docs/topics/practices.rst index 9ae34f423..60fe2267c 100644 --- a/docs/topics/practices.rst +++ b/docs/topics/practices.rst @@ -228,7 +228,7 @@ with varying degrees of sophistication. Getting around those measures can be difficult and tricky, and may sometimes require special infrastructure. Please consider contacting `commercial support`_ if in doubt. -Here are some tips to keep in mind when dealing with these kind of sites: +Here are some tips to keep in mind when dealing with these kinds of sites: * rotate your user agent from a pool of well-known ones from browsers (google around to get a list of them) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index 0abec1f96..ea64d1599 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -282,7 +282,7 @@ fields with form data from :class:`Response` objects. The :class:`FormRequest` objects support the following class method in addition to the standard :class:`Request` methods: - .. classmethod:: FormRequest.from_response(response, [formname=None, formnumber=0, formdata=None, formxpath=None, clickdata=None, dont_click=False, ...]) + .. classmethod:: FormRequest.from_response(response, [formname=None, formnumber=0, formdata=None, formxpath=None, formcss=None, clickdata=None, dont_click=False, ...]) Returns a new :class:`FormRequest` object with its form field values pre-populated with those found in the HTML ``
`` element contained @@ -310,6 +310,9 @@ fields with form data from :class:`Response` objects. :param formxpath: if given, the first form that matches the xpath will be used. :type formxpath: string + :param formcss: if given, the first form that matches the css selector will be used. + :type formcss: string + :param formnumber: the number of form to use, when the response contains multiple forms. The first one (and also the default) is ``0``. :type formnumber: integer @@ -339,6 +342,9 @@ fields with form data from :class:`Response` objects. .. versionadded:: 0.17 The ``formxpath`` parameter. + .. versionadded:: 1.1.0 + The ``formcss`` parameter. + Request usage examples ---------------------- diff --git a/docs/topics/selectors.rst b/docs/topics/selectors.rst index 273cae0f8..8dc82dfe5 100644 --- a/docs/topics/selectors.rst +++ b/docs/topics/selectors.rst @@ -579,7 +579,7 @@ Built-in Selectors reference is used together with ``text``. If ``type`` is ``None`` and a ``response`` is passed, the selector type is - inferred from the response type as follow: + inferred from the response type as follows: * ``"html"`` for :class:`~scrapy.http.HtmlResponse` type * ``"xml"`` for :class:`~scrapy.http.XmlResponse` type @@ -757,7 +757,7 @@ nodes can be accessed directly by their names:: >> response.xpath('//title/text()').extract_first() + u'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework' - >>> fetch("http://slashdot.org") + >>> fetch("http://reddit.com") [s] Available Scrapy objects: - [s] crawler + [s] crawler [s] item {} - [s] request - [s] response <200 http://slashdot.org> - [s] settings - [s] spider + [s] request + [s] response <200 https://www.reddit.com/> + [s] settings + [s] spider [s] Useful shortcuts: [s] shelp() Shell help (print this help) [s] fetch(req_or_url) Fetch request (or URL) and update local objects [s] view(response) View response in a browser >>> response.xpath('//title/text()').extract() - [u'Slashdot: News for nerds, stuff that matters'] + [u'reddit: the front page of the internet'] >>> request = request.replace(method="POST") diff --git a/docs/topics/spider-middleware.rst b/docs/topics/spider-middleware.rst index d448801d3..84daaaa55 100644 --- a/docs/topics/spider-middleware.rst +++ b/docs/topics/spider-middleware.rst @@ -24,20 +24,22 @@ Here's an example:: 'myproject.middlewares.CustomSpiderMiddleware': 543, } -The specified :setting:`SPIDER_MIDDLEWARES` setting is merged with the default -one (i.e. it does not overwrite it) and then sorted by order to get the final -sorted list of enabled middlewares: the first middleware is the one closer to -the engine and the last is the one closer to the spider. +The :setting:`SPIDER_MIDDLEWARES` setting is merged with the +:setting:`SPIDER_MIDDLEWARES_BASE` setting defined in Scrapy (and not meant to +be overridden) and then sorted by order to get the final sorted list of enabled +middlewares: the first middleware is the one closer to the engine and the last +is the one closer to the spider. -To decide which order to assign to your middleware see the default -:setting:`SPIDER_MIDDLEWARES` setting and pick a value according to where +To decide which order to assign to your middleware see the +:setting:`SPIDER_MIDDLEWARES_BASE` setting and pick a value according to where you want to insert the middleware. The order does matter because each middleware performs a different action and your middleware could depend on some previous (or subsequent) middleware being applied. -If you want to disable a builtin middleware you must define it in your project's -:setting:`SPIDER_MIDDLEWARES` setting and assign ``None`` as its value. For -example, if you want to disable the off-site middleware:: +If you want to disable a builtin middleware (the ones defined in +:setting:`SPIDER_MIDDLEWARES_BASE`, and enabled by default) you must define it +in your project :setting:`SPIDER_MIDDLEWARES` setting and assign `None` as its +value. For example, if you want to disable the off-site middleware:: SPIDER_MIDDLEWARES = { 'myproject.middlewares.CustomSpiderMiddleware': 543, @@ -171,7 +173,7 @@ information on how to use them and how to write your own spider middleware, see the :ref:`spider middleware usage guide `. For a list of the components enabled by default (and their orders) see the -:setting:`SPIDER_MIDDLEWARES` setting. +:setting:`SPIDER_MIDDLEWARES_BASE` setting. DepthMiddleware --------------- diff --git a/docs/topics/stats.rst b/docs/topics/stats.rst index 0837610d0..dd0c6216b 100644 --- a/docs/topics/stats.rst +++ b/docs/topics/stats.rst @@ -47,7 +47,7 @@ Set stat value:: Increment stat value:: - stats.inc_value('pages_crawled') + stats.inc_value('custom_count') Set stat value only if greater than previous:: @@ -59,13 +59,13 @@ Set stat value only if lower than previous:: Get stat value:: - >>> stats.get_value('pages_crawled') - 8 + >>> stats.get_value('custom_count') + 1 Get all stats:: >>> stats.get_stats() - {'pages_crawled': 1238, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)} + {'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)} Available Stats Collectors ========================== diff --git a/requirements-py3.txt b/requirements-py3.txt index a9a2e3be0..065095101 100644 --- a/requirements-py3.txt +++ b/requirements-py3.txt @@ -1,4 +1,4 @@ -Twisted >= 15.1.0 +Twisted >= 15.5.0 lxml>=3.2.4 pyOpenSSL>=0.13.1 cssselect>=0.9 diff --git a/scrapy/__init__.py b/scrapy/__init__.py index c0477f509..03ec6c667 100644 --- a/scrapy/__init__.py +++ b/scrapy/__init__.py @@ -2,7 +2,7 @@ Scrapy - a web crawling and web scraping framework written for Python """ -__all__ = ['__version__', 'version_info', 'optional_features', 'twisted_version', +__all__ = ['__version__', 'version_info', 'twisted_version', 'Spider', 'Request', 'FormRequest', 'Selector', 'Item', 'Field'] # Scrapy version @@ -27,15 +27,8 @@ del warnings from . import _monkeypatches del _monkeypatches -# WARNING: optional_features set is deprecated and will be removed soon. Do not use. -optional_features = set() -# TODO: backwards compatibility, remove for Scrapy 0.20 -optional_features.add('ssl') - from twisted import version as _txv twisted_version = (_txv.major, _txv.minor, _txv.micro) -if twisted_version >= (11, 1, 0): - optional_features.add('http11') # Declare top-level shortcuts from scrapy.spiders import Spider diff --git a/scrapy/_monkeypatches.py b/scrapy/_monkeypatches.py index 782891326..60e0de1f2 100644 --- a/scrapy/_monkeypatches.py +++ b/scrapy/_monkeypatches.py @@ -20,6 +20,6 @@ if sys.version_info[0] == 2: import twisted.persisted.styles # NOQA # Remove only entries with twisted serializers for non-twisted types. for k, v in frozenset(copyreg.dispatch_table.items()): - if not getattr(k, '__module__', '').startswith('twisted') \ - and getattr(v, '__module__', '').startswith('twisted'): + if not str(getattr(k, '__module__', '')).startswith('twisted') \ + and str(getattr(v, '__module__', '')).startswith('twisted'): copyreg.dispatch_table.pop(k) diff --git a/scrapy/cmdline.py b/scrapy/cmdline.py index 35050c13d..cb7bbd64d 100644 --- a/scrapy/cmdline.py +++ b/scrapy/cmdline.py @@ -7,7 +7,6 @@ import pkg_resources import scrapy from scrapy.crawler import CrawlerProcess -from scrapy.xlib import lsprofcalltree from scrapy.commands import ScrapyCommand from scrapy.exceptions import UsageError from scrapy.utils.misc import walk_modules @@ -144,7 +143,7 @@ def execute(argv=None, settings=None): sys.exit(cmd.exitcode) def _run_command(cmd, args, opts): - if opts.profile or opts.lsprof: + if opts.profile: _run_command_profiled(cmd, args, opts) else: cmd.run(args, opts) @@ -152,17 +151,11 @@ def _run_command(cmd, args, opts): def _run_command_profiled(cmd, args, opts): if opts.profile: sys.stderr.write("scrapy: writing cProfile stats to %r\n" % opts.profile) - if opts.lsprof: - sys.stderr.write("scrapy: writing lsprof stats to %r\n" % opts.lsprof) loc = locals() p = cProfile.Profile() p.runctx('cmd.run(args, opts)', globals(), loc) if opts.profile: p.dump_stats(opts.profile) - k = lsprofcalltree.KCacheGrind(p) - if opts.lsprof: - with open(opts.lsprof, 'w') as f: - k.output(f) if __name__ == '__main__': execute() diff --git a/scrapy/commands/__init__.py b/scrapy/commands/__init__.py index 9ac013098..43b420821 100644 --- a/scrapy/commands/__init__.py +++ b/scrapy/commands/__init__.py @@ -65,8 +65,6 @@ class ScrapyCommand(object): help="disable logging completely") group.add_option("--profile", metavar="FILE", default=None, help="write python cProfile stats to FILE") - group.add_option("--lsprof", metavar="FILE", default=None, - help="write lsprof profiling stats to FILE") group.add_option("--pidfile", metavar="FILE", help="write process ID to FILE") group.add_option("-s", "--set", action="append", default=[], metavar="NAME=VALUE", diff --git a/scrapy/commands/check.py b/scrapy/commands/check.py index a423ba2c9..b8a9ef989 100644 --- a/scrapy/commands/check.py +++ b/scrapy/commands/check.py @@ -58,7 +58,7 @@ class Command(ScrapyCommand): def run(self, args, opts): # load contracts - contracts = build_component_list(self.settings._getcomposite('SPIDER_CONTRACTS')) + contracts = build_component_list(self.settings.getwithbase('SPIDER_CONTRACTS')) conman = ContractsManager(load_object(c) for c in contracts) runner = TextTestRunner(verbosity=2 if opts.verbose else 1) result = TextTestResult(runner.stream, runner.descriptions, runner.verbosity) diff --git a/scrapy/commands/crawl.py b/scrapy/commands/crawl.py index 7f5c64c20..4b986bf9d 100644 --- a/scrapy/commands/crawl.py +++ b/scrapy/commands/crawl.py @@ -35,7 +35,8 @@ class Command(ScrapyCommand): self.settings.set('FEED_URI', 'stdout:', priority='cmdline') else: self.settings.set('FEED_URI', opts.output, priority='cmdline') - feed_exporters = without_none_values(self.settings._getcomposite('FEED_EXPORTERS')) + feed_exporters = without_none_values( + self.settings.getwithbase('FEED_EXPORTERS')) valid_output_formats = feed_exporters.keys() if not opts.output_format: opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") diff --git a/scrapy/commands/fetch.py b/scrapy/commands/fetch.py index e61eedf50..f09a873c1 100644 --- a/scrapy/commands/fetch.py +++ b/scrapy/commands/fetch.py @@ -1,4 +1,5 @@ from __future__ import print_function +import sys, six from w3lib.url import is_url from scrapy.commands import ScrapyCommand @@ -30,15 +31,19 @@ class Command(ScrapyCommand): def _print_headers(self, headers, prefix): for key, values in headers.items(): for value in values: - print('%s %s: %s' % (prefix, key, value)) + self._print_bytes(prefix + b' ' + key + b': ' + value) def _print_response(self, response, opts): if opts.headers: - self._print_headers(response.request.headers, '>') + self._print_headers(response.request.headers, b'>') print('>') - self._print_headers(response.headers, '<') + self._print_headers(response.headers, b'<') else: - print(response.body) + self._print_bytes(response.body) + + def _print_bytes(self, bytes_): + bytes_writer = sys.stdout if six.PY2 else sys.stdout.buffer + bytes_writer.write(bytes_ + b'\n') def run(self, args, opts): if len(args) != 1 or not is_url(args[0]): diff --git a/scrapy/commands/runspider.py b/scrapy/commands/runspider.py index 72229bcf5..1da09e4da 100644 --- a/scrapy/commands/runspider.py +++ b/scrapy/commands/runspider.py @@ -58,7 +58,7 @@ class Command(ScrapyCommand): self.settings.set('FEED_URI', 'stdout:', priority='cmdline') else: self.settings.set('FEED_URI', opts.output, priority='cmdline') - feed_exporters = without_none_values(self.settings._getcomposite('FEED_EXPORTERS')) + feed_exporters = without_none_values(self.settings.getwithbase('FEED_EXPORTERS')) valid_output_formats = feed_exporters.keys() if not opts.output_format: opts.output_format = os.path.splitext(opts.output)[1].replace(".", "") diff --git a/scrapy/commands/shell.py b/scrapy/commands/shell.py index 6f58a2300..14dd0a41a 100644 --- a/scrapy/commands/shell.py +++ b/scrapy/commands/shell.py @@ -19,7 +19,11 @@ from scrapy.utils.spider import spidercls_for_request, DefaultSpider class Command(ScrapyCommand): requires_project = False - default_settings = {'KEEP_ALIVE': True, 'LOGSTATS_INTERVAL': 0} + default_settings = { + 'KEEP_ALIVE': True, + 'LOGSTATS_INTERVAL': 0, + 'DUPEFILTER_CLASS': 'scrapy.dupefilters.BaseDupeFilter', + } def syntax(self): return "[url|file]" diff --git a/scrapy/commands/startproject.py b/scrapy/commands/startproject.py index 4375b6d4c..b3c5011df 100644 --- a/scrapy/commands/startproject.py +++ b/scrapy/commands/startproject.py @@ -1,10 +1,9 @@ from __future__ import print_function import re -import shutil import string from importlib import import_module from os.path import join, exists, abspath -from shutil import copytree, ignore_patterns +from shutil import copytree, ignore_patterns, move import scrapy from scrapy.commands import ScrapyCommand @@ -12,8 +11,6 @@ from scrapy.utils.template import render_templatefile, string_camelcase from scrapy.exceptions import UsageError -TEMPLATES_PATH = join(scrapy.__path__[0], 'templates', 'project') - TEMPLATES_TO_RENDER = ( ('scrapy.cfg',), ('${project_name}', 'settings.py.tmpl'), @@ -63,17 +60,24 @@ class Command(ScrapyCommand): self.exitcode = 1 return - moduletpl = join(TEMPLATES_PATH, 'module') - copytree(moduletpl, join(project_name, project_name), ignore=IGNORE) - shutil.copy(join(TEMPLATES_PATH, 'scrapy.cfg'), project_name) + copytree(self.templates_dir, project_name, ignore=IGNORE) + move(join(project_name, 'module'), join(project_name, project_name)) for paths in TEMPLATES_TO_RENDER: path = join(*paths) tplfile = join(project_name, string.Template(path).substitute(project_name=project_name)) render_templatefile(tplfile, project_name=project_name, ProjectName=string_camelcase(project_name)) - print("New Scrapy project %r created in:" % project_name) + print("New Scrapy project %r, using template directory %r, created in:" % \ + (project_name, self.templates_dir)) print(" %s\n" % abspath(project_name)) print("You can start your first spider with:") print(" cd %s" % project_name) print(" scrapy genspider example example.com") + + @property + def templates_dir(self): + _templates_base_dir = self.settings['TEMPLATES_DIR'] or \ + join(scrapy.__path__[0], 'templates') + return join(_templates_base_dir, 'project') + \ No newline at end of file diff --git a/scrapy/core/downloader/handlers/__init__.py b/scrapy/core/downloader/handlers/__init__.py index 0e78e04f4..bc5cd742e 100644 --- a/scrapy/core/downloader/handlers/__init__.py +++ b/scrapy/core/downloader/handlers/__init__.py @@ -20,7 +20,8 @@ class DownloadHandlers(object): self._schemes = {} # stores acceptable schemes on instancing self._handlers = {} # stores instanced handlers for schemes self._notconfigured = {} # remembers failed handlers - handlers = without_none_values(crawler.settings._getcomposite('DOWNLOAD_HANDLERS')) + handlers = without_none_values( + crawler.settings.getwithbase('DOWNLOAD_HANDLERS')) for scheme, clspath in six.iteritems(handlers): self._schemes[scheme] = clspath diff --git a/scrapy/core/downloader/handlers/http.py b/scrapy/core/downloader/handlers/http.py index 1efebb939..81da2615a 100644 --- a/scrapy/core/downloader/handlers/http.py +++ b/scrapy/core/downloader/handlers/http.py @@ -1,7 +1,7 @@ -from scrapy import optional_features +from scrapy import twisted_version from .http10 import HTTP10DownloadHandler -if 'http11' in optional_features: +if twisted_version >= (11, 1, 0): from .http11 import HTTP11DownloadHandler as HTTPDownloadHandler else: HTTPDownloadHandler = HTTP10DownloadHandler diff --git a/scrapy/core/downloader/handlers/http10.py b/scrapy/core/downloader/handlers/http10.py index 11b2acdae..0322bbe49 100644 --- a/scrapy/core/downloader/handlers/http10.py +++ b/scrapy/core/downloader/handlers/http10.py @@ -2,6 +2,7 @@ """ from twisted.internet import reactor from scrapy.utils.misc import load_object +from scrapy.utils.python import to_unicode class HTTP10DownloadHandler(object): @@ -17,8 +18,8 @@ class HTTP10DownloadHandler(object): return factory.deferred def _connect(self, factory): - host, port = factory.host, factory.port - if factory.scheme == 'https': + host, port = to_unicode(factory.host), factory.port + if factory.scheme == b'https': return reactor.connectSSL(host, port, factory, self.ClientContextFactory()) else: diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 31412a0f4..ad3285a32 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -6,7 +6,7 @@ from io import BytesIO from time import time from six.moves.urllib.parse import urldefrag -from zope.interface import implements +from zope.interface import implementer from twisted.internet import defer, reactor, protocol from twisted.web.http_headers import Headers as TxHeaders from twisted.web.iweb import IBodyProducer, UNKNOWN_LENGTH @@ -19,6 +19,7 @@ from scrapy.http import Headers from scrapy.responsetypes import responsetypes from scrapy.core.downloader.webclient import _parse from scrapy.utils.misc import load_object +from scrapy.utils.python import to_bytes, to_unicode from scrapy import twisted_version logger = logging.getLogger(__name__) @@ -77,7 +78,7 @@ class TunnelingTCP4ClientEndpoint(TCP4ClientEndpoint): for it. """ - _responseMatcher = re.compile('HTTP/1\.. 200') + _responseMatcher = re.compile(b'HTTP/1\.. 200') def __init__(self, reactor, host, port, proxyConf, contextFactory, timeout=30, bindAddress=None): @@ -91,11 +92,13 @@ class TunnelingTCP4ClientEndpoint(TCP4ClientEndpoint): def requestTunnel(self, protocol): """Asks the proxy to open a tunnel.""" - tunnelReq = 'CONNECT %s:%s HTTP/1.1\r\n' % (self._tunneledHost, - self._tunneledPort) + tunnelReq = to_bytes( + 'CONNECT %s:%s HTTP/1.1\r\n' % ( + self._tunneledHost, self._tunneledPort), encoding='ascii') if self._proxyAuthHeader: - tunnelReq += 'Proxy-Authorization: %s\r\n' % self._proxyAuthHeader - tunnelReq += '\r\n' + tunnelReq += \ + b'Proxy-Authorization: ' + self._proxyAuthHeader + b'\r\n' + tunnelReq += b'\r\n' protocol.transport.write(tunnelReq) self._protocolDataReceived = protocol.dataReceived protocol.dataReceived = self.processProxyResponse @@ -180,10 +183,11 @@ class ScrapyAgent(object): if proxy: _, _, proxyHost, proxyPort, proxyParams = _parse(proxy) scheme = _parse(request.url)[0] - omitConnectTunnel = proxyParams.find('noconnect') >= 0 - if scheme == 'https' and not omitConnectTunnel: + proxyHost = to_unicode(proxyHost) + omitConnectTunnel = b'noconnect' in proxyParams + if scheme == b'https' and not omitConnectTunnel: proxyConf = (proxyHost, proxyPort, - request.headers.get('Proxy-Authorization', None)) + request.headers.get(b'Proxy-Authorization', None)) return self._TunnelingAgent(reactor, proxyConf, contextFactory=self._contextFactory, connectTimeout=timeout, bindAddress=bindaddress, pool=self._pool) @@ -201,14 +205,15 @@ class ScrapyAgent(object): # request details url = urldefrag(request.url)[0] - method = request.method + method = to_bytes(request.method) headers = TxHeaders(request.headers) if isinstance(agent, self._TunnelingAgent): - headers.removeHeader('Proxy-Authorization') + headers.removeHeader(b'Proxy-Authorization') bodyproducer = _RequestBodyProducer(request.body) if request.body else None start_time = time() - d = agent.request(method, url, headers, bodyproducer) + d = agent.request( + method, to_bytes(url, encoding='ascii'), headers, bodyproducer) # set download latency d.addCallback(self._cb_latency, request, start_time) # response body is ready to be consumed @@ -232,18 +237,21 @@ class ScrapyAgent(object): def _cb_bodyready(self, txresponse, request): # deliverBody hangs for responses without body if txresponse.length == 0: - return txresponse, '', None + return txresponse, b'', None maxsize = request.meta.get('download_maxsize', self._maxsize) warnsize = request.meta.get('download_warnsize', self._warnsize) expected_size = txresponse.length if txresponse.length != UNKNOWN_LENGTH else -1 if maxsize and expected_size > maxsize: - logger.error("Expected response size (%(size)s) larger than " - "download max size (%(maxsize)s).", - {'size': expected_size, 'maxsize': maxsize}) + error_message = ("Cancelling download of {url}: expected response " + "size ({size}) larger than " + "download max size ({maxsize})." + ).format(url=request.url, size=expected_size, maxsize=maxsize) + + logger.error(error_message) txresponse._transport._producer.loseConnection() - raise defer.CancelledError() + raise defer.CancelledError(error_message) if warnsize and expected_size > warnsize: logger.warning("Expected response size (%(size)s) larger than " @@ -265,8 +273,8 @@ class ScrapyAgent(object): return respcls(url=url, status=status, headers=headers, body=body, flags=flags) +@implementer(IBodyProducer) class _RequestBodyProducer(object): - implements(IBodyProducer) def __init__(self, body): self.body = body @@ -292,6 +300,7 @@ class _ResponseReader(protocol.Protocol): self._bodybuf = BytesIO() self._maxsize = maxsize self._warnsize = warnsize + self._reached_warnsize = False self._bytes_received = 0 def dataReceived(self, bodyBytes): @@ -305,11 +314,12 @@ class _ResponseReader(protocol.Protocol): 'maxsize': self._maxsize}) self._finished.cancel() - if self._warnsize and self._bytes_received > self._warnsize: - logger.warning("Received (%(bytes)s) bytes larger than download " - "warn size (%(warnsize)s).", - {'bytes': self._bytes_received, - 'warnsize': self._warnsize}) + if self._warnsize and self._bytes_received > self._warnsize and not self._reached_warnsize: + self._reached_warnsize = True + logger.warning("Received more bytes than download " + "warn size (%(warnsize)s) in request %(request)s.", + {'warnsize': self._warnsize, + 'request': self._request}) def connectionLost(self, reason): if self._finished.called: diff --git a/scrapy/core/downloader/handlers/s3.py b/scrapy/core/downloader/handlers/s3.py index 311815b70..38cfd1e10 100644 --- a/scrapy/core/downloader/handlers/s3.py +++ b/scrapy/core/downloader/handlers/s3.py @@ -35,7 +35,7 @@ def get_s3_connection(): class S3DownloadHandler(object): def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None, \ - httpdownloadhandler=HTTPDownloadHandler): + httpdownloadhandler=HTTPDownloadHandler, **kw): _S3Connection = get_s3_connection() if _S3Connection is None: @@ -46,8 +46,15 @@ class S3DownloadHandler(object): if not aws_secret_access_key: aws_secret_access_key = settings['AWS_SECRET_ACCESS_KEY'] + # If no credentials could be found anywhere, + # consider this an anonymous connection request by default; + # unless 'anon' was set explicitly (True/False). + anon = kw.get('anon', None) + if anon is None and not aws_access_key_id and not aws_secret_access_key: + kw['anon'] = True + try: - self.conn = _S3Connection(aws_access_key_id, aws_secret_access_key) + self.conn = _S3Connection(aws_access_key_id, aws_secret_access_key, **kw) except Exception as ex: raise NotConfigured(str(ex)) self._download_http = httpdownloadhandler(settings).download_request diff --git a/scrapy/core/downloader/middleware.py b/scrapy/core/downloader/middleware.py index 958113fc3..c3b23e284 100644 --- a/scrapy/core/downloader/middleware.py +++ b/scrapy/core/downloader/middleware.py @@ -19,7 +19,8 @@ class DownloaderMiddlewareManager(MiddlewareManager): @classmethod def _get_mwlist_from_settings(cls, settings): - return build_component_list(settings._getcomposite('DOWNLOADER_MIDDLEWARES')) + return build_component_list( + settings.getwithbase('DOWNLOADER_MIDDLEWARES')) def _add_middleware(self, mw): if hasattr(mw, 'process_request'): diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index add5576ef..9bcc51943 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -7,21 +7,31 @@ from twisted.internet import defer from scrapy.http import Headers from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.python import to_bytes from scrapy.responsetypes import responsetypes def _parsed_url_args(parsed): + # Assume parsed is urlparse-d from Request.url, + # which was passed via safe_url_string and is ascii-only. + b = lambda s: to_bytes(s, encoding='ascii') path = urlunparse(('', '', parsed.path or '/', parsed.params, parsed.query, '')) - host = parsed.hostname + path = b(path) + host = b(parsed.hostname) port = parsed.port - scheme = parsed.scheme - netloc = parsed.netloc + scheme = b(parsed.scheme) + netloc = b(parsed.netloc) if port is None: - port = 443 if scheme == 'https' else 80 + port = 443 if scheme == b'https' else 80 return scheme, netloc, host, port, path def _parse(url): + """ Return tuple of (scheme, netloc, host, port, path), + all in bytes except for port which is int. + Assume url is from Request.url, which was passed via safe_url_string + and is ascii-only. + """ url = url.strip() parsed = urlparse(url) return _parsed_url_args(parsed) @@ -29,7 +39,7 @@ def _parse(url): class ScrapyHTTPPageGetter(HTTPClient): - delimiter = '\n' + delimiter = b'\n' def connectionMade(self): self.headers = Headers() # bucket for response headers @@ -63,8 +73,8 @@ class ScrapyHTTPPageGetter(HTTPClient): self.factory.noPage(reason) def handleResponse(self, response): - if self.factory.method.upper() == 'HEAD': - self.factory.page('') + if self.factory.method.upper() == b'HEAD': + self.factory.page(b'') elif self.length is not None and self.length > 0: self.factory.noPage(self._connection_lost_reason) else: @@ -91,8 +101,10 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): afterFoundGet = False def __init__(self, request, timeout=180): - self.url = urldefrag(request.url)[0] - self.method = request.method + self._url = urldefrag(request.url)[0] + # converting to bytes to comply to Twisted interface + self.url = to_bytes(self._url, encoding='ascii') + self.method = to_bytes(request.method, encoding='ascii') self.body = request.body or None self.headers = Headers(request.headers) self.response_headers = None @@ -119,15 +131,15 @@ class ScrapyHTTPClientFactory(HTTPClientFactory): # just in case a broken http/1.1 decides to keep connection alive self.headers.setdefault("Connection", "close") # Content-Length must be specified in POST method even with no body - elif self.method == 'POST': + elif self.method == b'POST': self.headers['Content-Length'] = 0 def _build_response(self, body, request): request.meta['download_latency'] = self.headers_time-self.start_time status = int(self.status) headers = Headers(self.response_headers) - respcls = responsetypes.from_args(headers=headers, url=self.url) - return respcls(url=self.url, status=status, headers=headers, body=body) + respcls = responsetypes.from_args(headers=headers, url=self._url) + return respcls(url=self._url, status=status, headers=headers, body=body) def _set_connection_attributes(self, request): parsed = urlparse_cached(request) diff --git a/scrapy/core/engine.py b/scrapy/core/engine.py index eb2779b12..ef4403106 100644 --- a/scrapy/core/engine.py +++ b/scrapy/core/engine.py @@ -7,7 +7,7 @@ For more information see docs/topics/architecture.rst import logging from time import time -from twisted.internet import defer +from twisted.internet import defer, task from twisted.python.failure import Failure from scrapy import signals @@ -30,6 +30,7 @@ class Slot(object): self.close_if_idle = close_if_idle self.nextcall = nextcall self.scheduler = scheduler + self.heartbeat = task.LoopingCall(nextcall.schedule) def add_request(self, request): self.inprogress.add(request) @@ -47,6 +48,7 @@ class Slot(object): if self.closing and not self.inprogress: if self.nextcall: self.nextcall.cancel() + self.heartbeat.stop() self.closing.callback(None) @@ -113,7 +115,6 @@ class ExecutionEngine(object): return if self.paused: - slot.nextcall.schedule(5) return while not self._needs_backout(spider): @@ -254,6 +255,7 @@ class ExecutionEngine(object): self.crawler.stats.open_spider(spider) yield self.signals.send_catch_log_deferred(signals.spider_opened, spider=spider) slot.nextcall.schedule() + slot.heartbeat.start(5) def _spider_idle(self, spider): """Called when a spider gets idle. This function is called when there @@ -267,7 +269,6 @@ class ExecutionEngine(object): spider=spider, dont_log=DontCloseSpider) if any(isinstance(x, Failure) and isinstance(x.value, DontCloseSpider) \ for _, x in res): - self.slot.nextcall.schedule(5) return if self.spider_is_idle(spider): diff --git a/scrapy/core/spidermw.py b/scrapy/core/spidermw.py index b5c80c350..a206e4b0c 100644 --- a/scrapy/core/spidermw.py +++ b/scrapy/core/spidermw.py @@ -18,7 +18,7 @@ class SpiderMiddlewareManager(MiddlewareManager): @classmethod def _get_mwlist_from_settings(cls, settings): - return build_component_list(settings._getcomposite('SPIDER_MIDDLEWARES')) + return build_component_list(settings.getwithbase('SPIDER_MIDDLEWARES')) def _add_middleware(self, mw): super(SpiderMiddlewareManager, self)._add_middleware(mw) diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 719507396..7ab304c17 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -9,13 +9,13 @@ from scrapy.exceptions import NotConfigured class HttpCompressionMiddleware(object): """This middleware allows compressed (gzip, deflate) traffic to be sent/received from web sites""" - + @classmethod def from_crawler(cls, crawler): if not crawler.settings.getbool('COMPRESSION_ENABLED'): raise NotConfigured return cls() - + def process_request(self, request, spider): request.headers.setdefault('Accept-Encoding', 'gzip,deflate') @@ -39,10 +39,10 @@ class HttpCompressionMiddleware(object): return response def _decode(self, body, encoding): - if encoding == 'gzip' or encoding == 'x-gzip': + if encoding == b'gzip' or encoding == b'x-gzip': body = gunzip(body) - if encoding == 'deflate': + if encoding == b'deflate': try: body = zlib.decompress(body) except zlib.error: diff --git a/scrapy/downloadermiddlewares/httpproxy.py b/scrapy/downloadermiddlewares/httpproxy.py index dda6a3d2a..b01bab76d 100644 --- a/scrapy/downloadermiddlewares/httpproxy.py +++ b/scrapy/downloadermiddlewares/httpproxy.py @@ -9,11 +9,13 @@ from six.moves.urllib.parse import urlunparse from scrapy.utils.httpobj import urlparse_cached from scrapy.exceptions import NotConfigured +from scrapy.utils.python import to_bytes class HttpProxyMiddleware(object): - def __init__(self): + def __init__(self, auth_encoding='latin-1'): + self.auth_encoding = auth_encoding self.proxies = {} for type, url in getproxies().items(): self.proxies[type] = self._get_proxy(url, type) @@ -21,12 +23,19 @@ class HttpProxyMiddleware(object): if not self.proxies: raise NotConfigured + @classmethod + def from_crawler(cls, crawler): + auth_encoding = crawler.settings.get('HTTPPROXY_AUTH_ENCODING') + return cls(auth_encoding) + def _get_proxy(self, url, orig_type): proxy_type, user, password, hostport = _parse_proxy(url) proxy_url = urlunparse((proxy_type or orig_type, hostport, '', '', '', '')) if user: - user_pass = '%s:%s' % (unquote(user), unquote(password)) + user_pass = to_bytes( + '%s:%s' % (unquote(user), unquote(password)), + encoding=self.auth_encoding) creds = base64.b64encode(user_pass).strip() else: creds = None @@ -52,4 +61,4 @@ class HttpProxyMiddleware(object): creds, proxy = self.proxies[scheme] request.meta['proxy'] = proxy if creds: - request.headers['Proxy-Authorization'] = 'Basic ' + creds + request.headers['Proxy-Authorization'] = b'Basic ' + creds diff --git a/scrapy/downloadermiddlewares/retry.py b/scrapy/downloadermiddlewares/retry.py index 803ed5fc0..3324aa21a 100644 --- a/scrapy/downloadermiddlewares/retry.py +++ b/scrapy/downloadermiddlewares/retry.py @@ -56,7 +56,7 @@ class RetryMiddleware(object): def process_exception(self, request, exception, spider): if isinstance(exception, self.EXCEPTIONS_TO_RETRY) \ and not request.meta.get('dont_retry', False): - return self._retry(request, exception, spider) + return self._retry(request, exception, spider) def _retry(self, request, reason, spider): retries = request.meta.get('retry_times', 0) + 1 diff --git a/scrapy/extension.py b/scrapy/extension.py index 4ceb32c68..e39e456fa 100644 --- a/scrapy/extension.py +++ b/scrapy/extension.py @@ -12,4 +12,4 @@ class ExtensionManager(MiddlewareManager): @classmethod def _get_mwlist_from_settings(cls, settings): - return build_component_list(settings._getcomposite('EXTENSIONS')) + return build_component_list(settings.getwithbase('EXTENSIONS')) diff --git a/scrapy/extensions/feedexport.py b/scrapy/extensions/feedexport.py index 1e27a1e7e..daea551cb 100644 --- a/scrapy/extensions/feedexport.py +++ b/scrapy/extensions/feedexport.py @@ -196,7 +196,7 @@ class FeedExporter(object): return item def _load_components(self, setting_prefix): - conf = without_none_values(self.settings._getcomposite(setting_prefix)) + conf = without_none_values(self.settings.getwithbase(setting_prefix)) d = {} for k, v in conf.items(): try: diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 2911dd6bc..a871cc895 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -12,6 +12,7 @@ from scrapy.responsetypes import responsetypes from scrapy.utils.request import request_fingerprint from scrapy.utils.project import data_path from scrapy.utils.httpobj import urlparse_cached +from scrapy.utils.python import to_bytes, to_unicode class DummyPolicy(object): @@ -40,12 +41,13 @@ class RFC2616Policy(object): def __init__(self, settings): self.always_store = settings.getbool('HTTPCACHE_ALWAYS_STORE') self.ignore_schemes = settings.getlist('HTTPCACHE_IGNORE_SCHEMES') - self.ignore_response_cache_controls = settings.getlist('HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS') + self.ignore_response_cache_controls = [to_bytes(cc) for cc in + settings.getlist('HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS')] self._cc_parsed = WeakKeyDictionary() def _parse_cachecontrol(self, r): if r not in self._cc_parsed: - cch = r.headers.get('Cache-Control', '') + cch = r.headers.get(b'Cache-Control', b'') parsed = parse_cachecontrol(cch) if isinstance(r, Response): for key in self.ignore_response_cache_controls: @@ -58,7 +60,7 @@ class RFC2616Policy(object): return False cc = self._parse_cachecontrol(request) # obey user-agent directive "Cache-Control: no-store" - if 'no-store' in cc: + if b'no-store' in cc: return False # Any other is eligible for caching return True @@ -69,7 +71,7 @@ class RFC2616Policy(object): # Status code 206 is not included because cache can not deal with partial contents cc = self._parse_cachecontrol(response) # obey directive "Cache-Control: no-store" - if 'no-store' in cc: + if b'no-store' in cc: return False # Never cache 304 (Not Modified) responses elif response.status == 304: @@ -78,14 +80,14 @@ class RFC2616Policy(object): elif self.always_store: return True # Any hint on response expiration is good - elif 'max-age' in cc or 'Expires' in response.headers: + elif b'max-age' in cc or b'Expires' in response.headers: return True # Firefox fallbacks this statuses to one year expiration if none is set elif response.status in (300, 301, 308): return True # Other statuses without expiration requires at least one validator elif response.status in (200, 203, 401): - return 'Last-Modified' in response.headers or 'ETag' in response.headers + return b'Last-Modified' in response.headers or b'ETag' in response.headers # Any other is probably not eligible for caching # Makes no sense to cache responses that does not contain expiration # info and can not be revalidated @@ -95,7 +97,7 @@ class RFC2616Policy(object): def is_cached_response_fresh(self, cachedresponse, request): cc = self._parse_cachecontrol(cachedresponse) ccreq = self._parse_cachecontrol(request) - if 'no-cache' in cc or 'no-cache' in ccreq: + if b'no-cache' in cc or b'no-cache' in ccreq: return False now = time() @@ -109,7 +111,7 @@ class RFC2616Policy(object): if currentage < freshnesslifetime: return True - if 'max-stale' in ccreq and 'must-revalidate' not in cc: + if b'max-stale' in ccreq and b'must-revalidate' not in cc: # From RFC2616: "Indicates that the client is willing to # accept a response that has exceeded its expiration time. # If max-stale is assigned a value, then the client is @@ -117,7 +119,7 @@ class RFC2616Policy(object): # expiration time by no more than the specified number of # seconds. If no value is assigned to max-stale, then the # client is willing to accept a stale response of any age." - staleage = ccreq['max-stale'] + staleage = ccreq[b'max-stale'] if staleage is None: return True @@ -136,22 +138,22 @@ class RFC2616Policy(object): # as long as the old response didn't specify must-revalidate. if response.status >= 500: cc = self._parse_cachecontrol(cachedresponse) - if 'must-revalidate' not in cc: + if b'must-revalidate' not in cc: return True # Use the cached response if the server says it hasn't changed. return response.status == 304 def _set_conditional_validators(self, request, cachedresponse): - if 'Last-Modified' in cachedresponse.headers: - request.headers['If-Modified-Since'] = cachedresponse.headers['Last-Modified'] + if b'Last-Modified' in cachedresponse.headers: + request.headers[b'If-Modified-Since'] = cachedresponse.headers[b'Last-Modified'] - if 'ETag' in cachedresponse.headers: - request.headers['If-None-Match'] = cachedresponse.headers['ETag'] + if b'ETag' in cachedresponse.headers: + request.headers[b'If-None-Match'] = cachedresponse.headers[b'ETag'] def _get_max_age(self, cc): try: - return max(0, int(cc['max-age'])) + return max(0, int(cc[b'max-age'])) except (KeyError, ValueError): return None @@ -164,18 +166,18 @@ class RFC2616Policy(object): return maxage # Parse date header or synthesize it if none exists - date = rfc1123_to_epoch(response.headers.get('Date')) or now + date = rfc1123_to_epoch(response.headers.get(b'Date')) or now # Try HTTP/1.0 Expires header - if 'Expires' in response.headers: - expires = rfc1123_to_epoch(response.headers['Expires']) + if b'Expires' in response.headers: + expires = rfc1123_to_epoch(response.headers[b'Expires']) # When parsing Expires header fails RFC 2616 section 14.21 says we # should treat this as an expiration time in the past. return max(0, expires - date) if expires else 0 # Fallback to heuristic using last-modified header # This is not in RFC but on Firefox caching implementation - lastmodified = rfc1123_to_epoch(response.headers.get('Last-Modified')) + lastmodified = rfc1123_to_epoch(response.headers.get(b'Last-Modified')) if lastmodified and lastmodified <= date: return (date - lastmodified) / 10 @@ -192,13 +194,13 @@ class RFC2616Policy(object): currentage = 0 # If Date header is not set we assume it is a fast connection, and # clock is in sync with the server - date = rfc1123_to_epoch(response.headers.get('Date')) or now + date = rfc1123_to_epoch(response.headers.get(b'Date')) or now if now > date: currentage = now - date - if 'Age' in response.headers: + if b'Age' in response.headers: try: - age = int(response.headers['Age']) + age = int(response.headers[b'Age']) currentage = max(currentage, age) except ValueError: pass @@ -305,7 +307,7 @@ class FilesystemCacheStorage(object): 'timestamp': time(), } with self._open(os.path.join(rpath, 'meta'), 'wb') as f: - f.write(repr(metadata)) + f.write(to_bytes(repr(metadata))) with self._open(os.path.join(rpath, 'pickled_meta'), 'wb') as f: pickle.dump(metadata, f, protocol=2) with self._open(os.path.join(rpath, 'response_headers'), 'wb') as f: @@ -373,14 +375,14 @@ class LeveldbCacheStorage(object): 'body': response.body, } batch = self._leveldb.WriteBatch() - batch.Put('%s_data' % key, pickle.dumps(data, protocol=2)) - batch.Put('%s_time' % key, str(time())) + batch.Put(key + b'_data', pickle.dumps(data, protocol=2)) + batch.Put(key + b'_time', to_bytes(str(time()))) self.db.Write(batch) def _read_data(self, spider, request): key = self._request_key(request) try: - ts = self.db.Get('%s_time' % key) + ts = self.db.Get(key + b'_time') except KeyError: return # not found or invalid entry @@ -388,14 +390,14 @@ class LeveldbCacheStorage(object): return # expired try: - data = self.db.Get('%s_data' % key) + data = self.db.Get(key + b'_data') except KeyError: return # invalid entry else: return pickle.loads(data) def _request_key(self, request): - return request_fingerprint(request) + return to_bytes(request_fingerprint(request)) @@ -404,16 +406,16 @@ def parse_cachecontrol(header): http://www.w3.org/Protocols/rfc2616/rfc2616-sec14.html#sec14.9 - >>> parse_cachecontrol('public, max-age=3600') == {'public': None, - ... 'max-age': '3600'} + >>> parse_cachecontrol(b'public, max-age=3600') == {b'public': None, + ... b'max-age': b'3600'} True - >>> parse_cachecontrol('') == {} + >>> parse_cachecontrol(b'') == {} True """ directives = {} - for directive in header.split(','): - key, sep, val = directive.strip().partition('=') + for directive in header.split(b','): + key, sep, val = directive.strip().partition(b'=') if key: directives[key.lower()] = val if sep else None return directives @@ -421,6 +423,7 @@ def parse_cachecontrol(header): def rfc1123_to_epoch(date_str): try: + date_str = to_unicode(date_str, encoding='ascii') return mktime_tz(parsedate_tz(date_str)) except Exception: return None diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py index a12a2fd07..5501634d3 100644 --- a/scrapy/http/request/form.py +++ b/scrapy/http/request/form.py @@ -11,6 +11,7 @@ from parsel.selector import create_root_node import six from scrapy.http.request import Request from scrapy.utils.python import to_bytes, is_listlike +from scrapy.utils.response import get_base_url class FormRequest(Request): @@ -33,8 +34,14 @@ class FormRequest(Request): @classmethod def from_response(cls, response, formname=None, formid=None, formnumber=0, formdata=None, - clickdata=None, dont_click=False, formxpath=None, **kwargs): + clickdata=None, dont_click=False, formxpath=None, formcss=None, **kwargs): + kwargs.setdefault('encoding', response.encoding) + + if formcss is not None: + from parsel.csstranslator import HTMLTranslator + formxpath = HTMLTranslator().css_to_xpath(formcss) + form = _get_form(response, formname, formid, formnumber, formxpath) formdata = _get_inputs(form, formdata, dont_click, clickdata, response) url = _get_form_url(form, kwargs.pop('url', None)) @@ -44,7 +51,7 @@ class FormRequest(Request): def _get_form_url(form, url): if url is None: - return form.action or form.base_url + return urljoin(form.base_url, form.action) return urljoin(form.base_url, url) @@ -58,7 +65,7 @@ def _urlencode(seq, enc): def _get_form(response, formname, formid, formnumber, formxpath): """Find the form element """ text = response.body_as_unicode() - root = create_root_node(text, lxml.html.HTMLParser, base_url=response.url) + root = create_root_node(text, lxml.html.HTMLParser, base_url=get_base_url(response)) forms = root.xpath('//form') if not forms: raise ValueError("No element found in %s" % response) @@ -72,7 +79,7 @@ def _get_form(response, formname, formid, formnumber, formxpath): f = root.xpath('//form[@id="%s"]' % formid) if f: return f[0] - + # Get form element from xpath, if not found, go up if formxpath is not None: nodes = root.xpath(formxpath) @@ -84,7 +91,8 @@ def _get_form(response, formname, formid, formnumber, formxpath): el = el.getparent() if el is None: break - raise ValueError('No element found with %s' % formxpath) + encoded = formxpath if six.PY3 else formxpath.encode('unicode_escape') + raise ValueError('No element found with %s' % encoded) # If we get here, it means that either formname was None # or invalid @@ -106,8 +114,12 @@ def _get_inputs(form, formdata, dont_click, clickdata, response): inputs = form.xpath('descendant::textarea' '|descendant::select' - '|descendant::input[@type!="submit" and @type!="image" and @type!="reset"' - 'and ((@type!="checkbox" and @type!="radio") or @checked)]') + '|descendant::input[not(@type) or @type[' + ' not(re:test(., "^(?:submit|image|reset)$", "i"))' + ' and (../@checked or' + ' not(re:test(., "^(?:checkbox|radio)$", "i")))]]', + namespaces={ + "re": "http://exslt.org/regular-expressions"}) values = [(k, u'' if v is None else v) for k, v in (_value(e) for e in inputs) if k and k not in formdata] @@ -150,9 +162,13 @@ def _get_clickable(clickdata, form): if the latter is given. If not, it returns the first clickable element found """ - clickables = [el for el in form.xpath('descendant::input[@type="submit"]' - '|descendant::button[@type="submit"]' - '|descendant::button[not(@type)]')] + clickables = [ + el for el in form.xpath( + 'descendant::*[(self::input or self::button)' + ' and re:test(@type, "^submit$", "i")]' + '|descendant::button[not(@type)]', + namespaces={"re": "http://exslt.org/regular-expressions"}) + ] if not clickables: return diff --git a/scrapy/linkextractors/regex.py b/scrapy/linkextractors/regex.py index b6f8d5d30..0fc7b079f 100644 --- a/scrapy/linkextractors/regex.py +++ b/scrapy/linkextractors/regex.py @@ -1,7 +1,7 @@ import re from six.moves.urllib.parse import urljoin -from w3lib.html import remove_tags, replace_entities, replace_escape_chars +from w3lib.html import remove_tags, replace_entities, replace_escape_chars, get_base_url from scrapy.link import Link from .sgml import SgmlLinkExtractor @@ -31,7 +31,7 @@ class RegexLinkExtractor(SgmlLinkExtractor): return clean_url if base_url is None: - base_url = urljoin(response_url, self.base_url) if self.base_url else response_url + base_url = get_base_url(response_text, response_url, response_encoding) links_text = linkre.findall(response_text) return [Link(clean_url(url).encode(response_encoding), diff --git a/scrapy/linkextractors/sgml.py b/scrapy/linkextractors/sgml.py index 7084d0180..9938e071f 100644 --- a/scrapy/linkextractors/sgml.py +++ b/scrapy/linkextractors/sgml.py @@ -124,9 +124,6 @@ class SgmlLinkExtractor(FilteringLinkExtractor): restrict_xpaths=restrict_xpaths, restrict_css=restrict_css, canonicalize=canonicalize, deny_extensions=deny_extensions) - # FIXME: was added to fix a RegexLinkExtractor testcase - self.base_url = None - def extract_links(self, response): base_url = None if self.restrict_xpaths: diff --git a/scrapy/pipelines/__init__.py b/scrapy/pipelines/__init__.py index 8df0d3154..2ef8786d0 100644 --- a/scrapy/pipelines/__init__.py +++ b/scrapy/pipelines/__init__.py @@ -13,7 +13,7 @@ class ItemPipelineManager(MiddlewareManager): @classmethod def _get_mwlist_from_settings(cls, settings): - return build_component_list(settings._getcomposite('ITEM_PIPELINES')) + return build_component_list(settings.getwithbase('ITEM_PIPELINES')) def _add_middleware(self, pipe): super(ItemPipelineManager, self)._add_middleware(pipe) diff --git a/scrapy/settings/__init__.py b/scrapy/settings/__init__.py index 13656298b..342d2585e 100644 --- a/scrapy/settings/__init__.py +++ b/scrapy/settings/__init__.py @@ -49,14 +49,11 @@ class SettingsAttribute(object): def set(self, value, priority): """Sets value if priority is higher or equal than current priority.""" - if isinstance(self.value, BaseSettings): - # Ignore self.priority if self.value has per-key priorities - self.value.update(value, priority) - self.priority = max(self.value.maxpriority(), priority) - else: - if priority >= self.priority: - self.value = value - self.priority = priority + if priority >= self.priority: + if isinstance(self.value, BaseSettings): + value = BaseSettings(value, priority=priority) + self.value = value + self.priority = priority def __str__(self): return " get_settings_priority('default'): - compsett.set(k, self[name][k], prio) - return compsett - return self[name] + def getwithbase(self, name): + """Get a composition of a dictionary-like setting and its `_BASE` + counterpart. + + :param name: name of the dictionary-like setting + :type name: string + """ + compbs = BaseSettings() + compbs.update(self[name + '_BASE']) + compbs.update(self[name]) + return compbs def getpriority(self, name): """ @@ -223,10 +211,9 @@ class BaseSettings(MutableMapping): :param name: the setting name :type name: string """ - prio = None - if name in self: - prio = self.attributes[name].priority - return prio + if name not in self: + return None + return self.attributes[name].priority def maxpriority(self): """ diff --git a/scrapy/settings/default_settings.py b/scrapy/settings/default_settings.py index 375efcdbb..44e74dc61 100644 --- a/scrapy/settings/default_settings.py +++ b/scrapy/settings/default_settings.py @@ -18,6 +18,8 @@ import sys from importlib import import_module from os.path import join, abspath, dirname +import six + AJAXCRAWL_ENABLED = False AUTOTHROTTLE_ENABLED = False @@ -63,7 +65,8 @@ DNS_TIMEOUT = 60 DOWNLOAD_DELAY = 0 -DOWNLOAD_HANDLERS = { +DOWNLOAD_HANDLERS = {} +DOWNLOAD_HANDLERS_BASE = { 'file': 'scrapy.core.downloader.handlers.file.FileDownloadHandler', 'http': 'scrapy.core.downloader.handlers.http.HTTPDownloadHandler', 'https': 'scrapy.core.downloader.handlers.http.HTTPDownloadHandler', @@ -81,7 +84,9 @@ DOWNLOADER = 'scrapy.core.downloader.Downloader' DOWNLOADER_HTTPCLIENTFACTORY = 'scrapy.core.downloader.webclient.ScrapyHTTPClientFactory' DOWNLOADER_CLIENTCONTEXTFACTORY = 'scrapy.core.downloader.contextfactory.ScrapyClientContextFactory' -DOWNLOADER_MIDDLEWARES = { +DOWNLOADER_MIDDLEWARES = {} + +DOWNLOADER_MIDDLEWARES_BASE = { # Engine side 'scrapy.downloadermiddlewares.robotstxt.RobotsTxtMiddleware': 100, 'scrapy.downloadermiddlewares.httpauth.HttpAuthMiddleware': 300, @@ -113,7 +118,9 @@ except KeyError: else: EDITOR = 'vi' -EXTENSIONS = { +EXTENSIONS = {} + +EXTENSIONS_BASE = { 'scrapy.extensions.corestats.CoreStats': 0, 'scrapy.extensions.telnet.TelnetConsole': 0, 'scrapy.extensions.memusage.MemoryUsage': 0, @@ -130,14 +137,16 @@ FEED_URI_PARAMS = None # a function to extend uri arguments FEED_FORMAT = 'jsonlines' FEED_STORE_EMPTY = False FEED_EXPORT_FIELDS = None -FEED_STORAGES = { +FEED_STORAGES = {} +FEED_STORAGES_BASE = { '': 'scrapy.extensions.feedexport.FileFeedStorage', 'file': 'scrapy.extensions.feedexport.FileFeedStorage', 'stdout': 'scrapy.extensions.feedexport.StdoutFeedStorage', 's3': 'scrapy.extensions.feedexport.S3FeedStorage', 'ftp': 'scrapy.extensions.feedexport.FTPFeedStorage', } -FEED_EXPORTERS = { +FEED_EXPORTERS = {} +FEED_EXPORTERS_BASE = { 'json': 'scrapy.exporters.JsonItemExporter', 'jsonlines': 'scrapy.exporters.JsonLinesItemExporter', 'jl': 'scrapy.exporters.JsonLinesItemExporter', @@ -156,13 +165,16 @@ HTTPCACHE_ALWAYS_STORE = False HTTPCACHE_IGNORE_HTTP_CODES = [] HTTPCACHE_IGNORE_SCHEMES = ['file'] HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS = [] -HTTPCACHE_DBM_MODULE = 'anydbm' +HTTPCACHE_DBM_MODULE = 'anydbm' if six.PY2 else 'dbm' HTTPCACHE_POLICY = 'scrapy.extensions.httpcache.DummyPolicy' HTTPCACHE_GZIP = False +HTTPPROXY_AUTH_ENCODING = 'latin-1' + ITEM_PROCESSOR = 'scrapy.pipelines.ItemPipelineManager' ITEM_PIPELINES = {} +ITEM_PIPELINES_BASE = {} LOG_ENABLED = True LOG_ENCODING = 'utf-8' @@ -221,7 +233,9 @@ SCHEDULER_MEMORY_QUEUE = 'scrapy.squeues.LifoMemoryQueue' SPIDER_LOADER_CLASS = 'scrapy.spiderloader.SpiderLoader' -SPIDER_MIDDLEWARES = { +SPIDER_MIDDLEWARES = {} + +SPIDER_MIDDLEWARES_BASE = { # Engine side 'scrapy.spidermiddlewares.httperror.HttpErrorMiddleware': 50, 'scrapy.spidermiddlewares.offsite.OffsiteMiddleware': 500, @@ -248,7 +262,8 @@ TELNETCONSOLE_ENABLED = 1 TELNETCONSOLE_PORT = [6023, 6073] TELNETCONSOLE_HOST = '127.0.0.1' -SPIDER_CONTRACTS = { +SPIDER_CONTRACTS = {} +SPIDER_CONTRACTS_BASE = { 'scrapy.contracts.default.UrlContract': 1, 'scrapy.contracts.default.ReturnsContract': 2, 'scrapy.contracts.default.ScrapesContract': 3, diff --git a/scrapy/squeues.py b/scrapy/squeues.py index 6e2a60fd2..21520f454 100644 --- a/scrapy/squeues.py +++ b/scrapy/squeues.py @@ -25,7 +25,9 @@ def _serializable_queue(queue_class, serialize, deserialize): def _pickle_serialize(obj): try: return pickle.dumps(obj, protocol=2) - except pickle.PicklingError as e: + # Python>=3.5 raises AttributeError here while + # Python<=3.4 raises pickle.PicklingError + except (pickle.PicklingError, AttributeError) as e: raise ValueError(str(e)) PickleFifoDiskQueue = _serializable_queue(queue.FifoDiskQueue, \ diff --git a/scrapy/utils/benchserver.py b/scrapy/utils/benchserver.py index 4385d72a9..5bbda6e27 100644 --- a/scrapy/utils/benchserver.py +++ b/scrapy/utils/benchserver.py @@ -13,18 +13,18 @@ class Root(Resource): return self def render(self, request): - total = _getarg(request, 'total', 100, int) - show = _getarg(request, 'show', 10, int) + total = _getarg(request, b'total', 100, int) + show = _getarg(request, b'show', 10, int) nlist = [random.randint(1, total) for _ in range(show)] - request.write("") + request.write(b"") args = request.args.copy() for nl in nlist: args['n'] = nl argstr = urlencode(args, doseq=True) request.write("follow {1}
" - .format(argstr, nl)) - request.write("") - return '' + .format(argstr, nl).encode('utf8')) + request.write(b"") + return b'' def _getarg(request, name, default=None, type=str): diff --git a/scrapy/utils/conf.py b/scrapy/utils/conf.py index 57f2b6322..e8af90f11 100644 --- a/scrapy/utils/conf.py +++ b/scrapy/utils/conf.py @@ -10,7 +10,7 @@ from scrapy.utils.deprecate import update_classpath from scrapy.utils.python import without_none_values -def build_component_list(compdict, convert=update_classpath): +def build_component_list(compdict, custom=None, convert=update_classpath): """Compose a component list from a { class: order } dictionary.""" def _check_components(complist): @@ -34,9 +34,15 @@ def build_component_list(compdict, convert=update_classpath): _check_components(compdict) return {convert(k): v for k, v in six.iteritems(compdict)} - if isinstance(compdict, (list, tuple)): - _check_components(compdict) - return type(compdict)(convert(c) for c in compdict) + # BEGIN Backwards compatibility for old (base, custom) call signature + if isinstance(custom, (list, tuple)): + _check_components(custom) + return type(custom)(convert(c) for c in custom) + + if custom is not None: + compdict.update(custom) + # END Backwards compatibility + compdict = without_none_values(_map_keys(compdict)) return [k for k, v in sorted(six.iteritems(compdict), key=itemgetter(1))] diff --git a/scrapy/utils/gz.py b/scrapy/utils/gz.py index 7fa4bba57..d035f9fdf 100644 --- a/scrapy/utils/gz.py +++ b/scrapy/utils/gz.py @@ -4,9 +4,25 @@ try: from cStringIO import StringIO as BytesIO except ImportError: from io import BytesIO - from gzip import GzipFile +import six + + +# - Python>=3.5 GzipFile's read() has issues returning leftover +# uncompressed data when input is corrupted +# (regression or bug-fix compared to Python 3.4) +# - read1(), which fetches data before raising EOFError on next call +# works here but is only available from Python>=3.3 +# - scrapy does not support Python 3.2 +# - Python 2.7 GzipFile works fine with standard read() + extrabuf +if six.PY2: + def read1(gzf, size=-1): + return gzf.read(size) +else: + def read1(gzf, size=-1): + return gzf.read1(size) + def gunzip(data): """Gunzip the given data and return as much data as possible. @@ -18,16 +34,18 @@ def gunzip(data): chunk = b'.' while chunk: try: - chunk = f.read(8196) + chunk = read1(f, 8196) output += chunk except (IOError, EOFError, struct.error): # complete only if there is some data, otherwise re-raise # see issue 87 about catching struct.error # some pages are quite small so output is '' and f.extrabuf # contains the whole page content - if output or f.extrabuf: - output += f.extrabuf - break + if output or getattr(f, 'extrabuf', None): + try: + output += f.extrabuf + finally: + break else: raise return output diff --git a/scrapy/utils/iterators.py b/scrapy/utils/iterators.py index c0d93f7a9..b0688791e 100644 --- a/scrapy/utils/iterators.py +++ b/scrapy/utils/iterators.py @@ -1,12 +1,11 @@ import re import csv import logging - try: from cStringIO import StringIO as BytesIO except ImportError: from io import BytesIO - +from io import StringIO import six from scrapy.http import TextResponse, Response @@ -18,7 +17,7 @@ logger = logging.getLogger(__name__) def xmliter(obj, nodename): """Return a iterator of Selector's over all nodes of a XML document, - given tha name of the node to iterate. Useful for parsing XML feeds. + given the name of the node to iterate. Useful for parsing XML feeds. obj can be: - a Response object @@ -36,7 +35,7 @@ def xmliter(obj, nodename): header_end = re_rsearch(HEADER_END_RE, text) header_end = text[header_end[1]:].strip() if header_end else '' - r = re.compile(r"<{0}[\s>].*?".format(nodename_patt), re.DOTALL) + r = re.compile(r'<%(np)s[\s>].*?' % {'np': nodename_patt}, re.DOTALL) for match in r.finditer(text): nodetext = header_start + match.group() + header_end yield Selector(text=nodetext, type='xml').xpath('//' + nodename)[0] @@ -49,7 +48,7 @@ def xmliter_lxml(obj, nodename, namespace=None, prefix='x'): iterable = etree.iterparse(reader, tag=tag, encoding=reader.encoding) selxpath = '//' + ('%s:%s' % (prefix, nodename) if namespace else nodename) for _, node in iterable: - nodetext = etree.tostring(node) + nodetext = etree.tostring(node, encoding='unicode') node.clear() xs = Selector(text=nodetext, type='xml') if namespace: @@ -65,7 +64,7 @@ class _StreamReader(object): self._text, self.encoding = obj.body, obj.encoding else: self._text, self.encoding = obj, 'utf-8' - self._is_unicode = isinstance(self._text, unicode) + self._is_unicode = isinstance(self._text, six.text_type) def read(self, n=65535): self.read = self._read_unicode if self._is_unicode else self._read_string @@ -94,7 +93,7 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): headers is an iterable that when provided offers the keys for the returned dictionaries, if not the first row is used. - + quotechar is the character used to enclosure fields on the given obj. """ @@ -102,7 +101,11 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): def _getrow(csv_r): return [to_unicode(field, encoding) for field in next(csv_r)] - lines = BytesIO(_body_or_str(obj, unicode=False)) + # Python 3 csv reader input object needs to return strings + if six.PY3: + lines = StringIO(_body_or_str(obj, unicode=True)) + else: + lines = BytesIO(_body_or_str(obj, unicode=False)) kwargs = {} if delimiter: kwargs["delimiter"] = delimiter @@ -125,8 +128,11 @@ def csviter(obj, delimiter=None, headers=None, encoding=None, quotechar=None): def _body_or_str(obj, unicode=True): - assert isinstance(obj, (Response, six.string_types)), \ - "obj must be Response or basestring, not %s" % type(obj).__name__ + expected_types = (Response, six.text_type, six.binary_type) + assert isinstance(obj, expected_types), \ + "obj must be %s, not %s" % ( + " or ".join(t.__name__ for t in expected_types), + type(obj).__name__) if isinstance(obj, Response): if not unicode: return obj.body diff --git a/scrapy/utils/log.py b/scrapy/utils/log.py index d40202953..cc2f0b164 100644 --- a/scrapy/utils/log.py +++ b/scrapy/utils/log.py @@ -126,9 +126,6 @@ def log_scrapy_info(settings): logger.info("Scrapy %(version)s started (bot: %(bot)s)", {'version': scrapy.__version__, 'bot': settings['BOT_NAME']}) - logger.info("Optional features available: %(features)s", - {'features': ", ".join(scrapy.optional_features)}) - d = dict(overridden_settings(settings)) logger.info("Overridden settings: %(settings)r", {'settings': d}) diff --git a/scrapy/utils/testsite.py b/scrapy/utils/testsite.py index 01508bdb4..ad0375443 100644 --- a/scrapy/utils/testsite.py +++ b/scrapy/utils/testsite.py @@ -22,13 +22,13 @@ class SiteTest(object): def test_site(): r = resource.Resource() - r.putChild("text", static.Data("Works", "text/plain")) - r.putChild("html", static.Data("

Works

World

", "text/html")) - r.putChild("enc-gb18030", static.Data("

gb18030 encoding

", "text/html; charset=gb18030")) - r.putChild("redirect", util.Redirect("/redirected")) - r.putChild("redirected", static.Data("Redirected here", "text/plain")) + r.putChild(b"text", static.Data(b"Works", "text/plain")) + r.putChild(b"html", static.Data(b"

Works

World

", "text/html")) + r.putChild(b"enc-gb18030", static.Data(b"

gb18030 encoding

", "text/html; charset=gb18030")) + r.putChild(b"redirect", util.Redirect(b"/redirected")) + r.putChild(b"redirected", static.Data(b"Redirected here", "text/plain")) return server.Site(r) - + if __name__ == '__main__': port = reactor.listenTCP(0, test_site(), interface="127.0.0.1") diff --git a/scrapy/xlib/lsprofcalltree.py b/scrapy/xlib/lsprofcalltree.py deleted file mode 100644 index a604016cc..000000000 --- a/scrapy/xlib/lsprofcalltree.py +++ /dev/null @@ -1,120 +0,0 @@ -# lsprofcalltree.py: lsprof output which is readable by kcachegrind -# David Allouche -# Jp Calderone & Itamar Shtull-Trauring -# Johan Dahlin - -from __future__ import print_function -import optparse -import os -import sys - -try: - import cProfile -except ImportError: - raise SystemExit("This script requires cProfile from Python 2.5") - -def label(code): - if isinstance(code, str): - return ('~', 0, code) # built-in functions ('~' sorts at the end) - else: - return '%s %s:%d' % (code.co_name, - code.co_filename, - code.co_firstlineno) - -class KCacheGrind(object): - def __init__(self, profiler): - self.data = profiler.getstats() - self.out_file = None - - def output(self, out_file): - self.out_file = out_file - print('events: Ticks', file=out_file) - self._print_summary() - for entry in self.data: - self._entry(entry) - - def _print_summary(self): - max_cost = 0 - for entry in self.data: - totaltime = int(entry.totaltime * 1000) - max_cost = max(max_cost, totaltime) - print('summary: %d' % (max_cost,), file=self.out_file) - - def _entry(self, entry): - out_file = self.out_file - - code = entry.code - #print >> out_file, 'ob=%s' % (code.co_filename,) - if isinstance(code, str): - print('fi=~', file=out_file) - else: - print('fi=%s' % (code.co_filename,), file=out_file) - print('fn=%s' % (label(code),), file=out_file) - - inlinetime = int(entry.inlinetime * 1000) - if isinstance(code, str): - print('0 ', inlinetime, file=out_file) - else: - print('%d %d' % (code.co_firstlineno, inlinetime), file=out_file) - - # recursive calls are counted in entry.calls - if entry.calls: - calls = entry.calls - else: - calls = [] - - if isinstance(code, str): - lineno = 0 - else: - lineno = code.co_firstlineno - - for subentry in calls: - self._subentry(lineno, subentry) - print(file=out_file) - - def _subentry(self, lineno, subentry): - out_file = self.out_file - code = subentry.code - #print >> out_file, 'cob=%s' % (code.co_filename,) - print('cfn=%s' % (label(code),), file=out_file) - if isinstance(code, str): - print('cfi=~', file=out_file) - print('calls=%d 0' % (subentry.callcount,), file=out_file) - else: - print('cfi=%s' % (code.co_filename,), file=out_file) - print('calls=%d %d' % ( - subentry.callcount, code.co_firstlineno), file=out_file) - - totaltime = int(subentry.totaltime * 1000) - print('%d %d' % (lineno, totaltime), file=out_file) - -def main(args): - usage = "%s [-o output_file_path] scriptfile [arg] ..." - parser = optparse.OptionParser(usage=usage % sys.argv[0]) - parser.allow_interspersed_args = False - parser.add_option('-o', '--outfile', dest="outfile", - help="Save stats to ", default=None) - - if not sys.argv[1:]: - parser.print_usage() - sys.exit(2) - - options, args = parser.parse_args() - - if not options.outfile: - options.outfile = '%s.log' % os.path.basename(args[0]) - - sys.argv[:] = args - - prof = cProfile.Profile() - try: - try: - prof = prof.run('execfile(%r)' % (sys.argv[0],)) - except SystemExit: - pass - finally: - kg = KCacheGrind(prof) - kg.output(file(options.outfile, 'w')) - -if __name__ == '__main__': - sys.exit(main(sys.argv)) diff --git a/sep/sep-003.rst b/sep/sep-003.rst index 282574968..184839525 100644 --- a/sep/sep-003.rst +++ b/sep/sep-003.rst @@ -146,7 +146,7 @@ Default values p['numbers'] # returns [] -Accesing and changing nested item values +Accessing and changing nested item values ---------------------------------------- :: diff --git a/sep/sep-014.rst b/sep/sep-014.rst index 98a31b1aa..8ca81824d 100644 --- a/sep/sep-014.rst +++ b/sep/sep-014.rst @@ -54,7 +54,7 @@ Request Extractors Request Extractors takes response object and determines which requests follow. -This is an enhancemente to ``LinkExtractors`` which returns urls (links), +This is an enhancement to ``LinkExtractors`` which returns urls (links), Request Extractors return Request objects. Request Processors diff --git a/sep/sep-018.rst b/sep/sep-018.rst index e30821917..aca7ac342 100644 --- a/sep/sep-018.rst +++ b/sep/sep-018.rst @@ -477,7 +477,7 @@ This is a port of the Offsite middleware to the new spider middleware API: def should_follow(self, request, spider): info = self.spiders[spider] - # hostanme can be None for wrong urls (like javascript links) + # hostname can be None for wrong urls (like javascript links) host = urlparse_cached(request).hostname or '' return bool(info.regex.search(host)) diff --git a/sep/sep-020.rst b/sep/sep-020.rst index 7b2c043b7..49d068479 100644 --- a/sep/sep-020.rst +++ b/sep/sep-020.rst @@ -23,9 +23,9 @@ Rationale ========= There are certain markup patterns that lend themselves quite nicely to -automated parsing, for example the ```` tag outlilnes such a pattern +automated parsing, for example the ``
`` tag outlines such a pattern for populating a database table with the embedded ```` elements denoting -the rows and the furthur embedded ``
`` elements denoting the individual +the rows and the further embedded ```` elements denoting the individual fields. One pattern that is particularly well suited for auto-populating an Item Loader diff --git a/tests/mockserver.py b/tests/mockserver.py index 3e4f8c0e5..365ec81fd 100644 --- a/tests/mockserver.py +++ b/tests/mockserver.py @@ -1,12 +1,12 @@ from __future__ import print_function import sys, time, random, os, json -import six from six.moves.urllib.parse import urlencode from subprocess import Popen, PIPE from twisted.web.server import Site, NOT_DONE_YET from twisted.web.resource import Resource from twisted.internet import reactor, defer, ssl from scrapy import twisted_version +from scrapy.utils.python import to_bytes, to_unicode if twisted_version < (11, 0, 0): @@ -30,9 +30,12 @@ else: from twisted.internet.task import deferLater -def getarg(request, name, default=None, type=str): +def getarg(request, name, default=None, type=None): if name in request.args: - return type(request.args[name][0]) + value = request.args[name][0] + if type is not None: + value = type(value) + return value else: return default @@ -55,12 +58,12 @@ class LeafResource(Resource): class Follow(LeafResource): def render(self, request): - total = getarg(request, "total", 100, type=int) - show = getarg(request, "show", 1, type=int) - order = getarg(request, "order", "desc") - maxlatency = getarg(request, "maxlatency", 0, type=float) - n = getarg(request, "n", total, type=int) - if order == "rand": + total = getarg(request, b"total", 100, type=int) + show = getarg(request, b"show", 1, type=int) + order = getarg(request, b"order", b"desc") + maxlatency = getarg(request, b"maxlatency", 0, type=float) + n = getarg(request, b"n", total, type=int) + if order == b"rand": nlist = [random.randint(1, total) for _ in range(show)] else: # order == "desc" nlist = range(n, max(n - show, 0), -1) @@ -73,19 +76,19 @@ class Follow(LeafResource): s = """ """ args = request.args.copy() for nl in nlist: - args["n"] = [str(nl)] + args[b"n"] = [to_bytes(str(nl))] argstr = urlencode(args, doseq=True) s += "follow %d
" % (argstr, nl) s += """""" - request.write(s) + request.write(to_bytes(s)) request.finish() class Delay(LeafResource): def render_GET(self, request): - n = getarg(request, "n", 1, type=float) - b = getarg(request, "b", 1, type=int) + n = getarg(request, b"n", 1, type=float) + b = getarg(request, b"b", 1, type=int) if b: # send headers now and delay body request.write('') @@ -93,16 +96,16 @@ class Delay(LeafResource): return NOT_DONE_YET def _delayedRender(self, request, n): - request.write("Response delayed for %0.3f seconds\n" % n) + request.write(to_bytes("Response delayed for %0.3f seconds\n" % n)) request.finish() class Status(LeafResource): def render_GET(self, request): - n = getarg(request, "n", 200, type=int) + n = getarg(request, b"n", 200, type=int) request.setResponseCode(n) - return "" + return b"" class Raw(LeafResource): @@ -114,7 +117,7 @@ class Raw(LeafResource): render_POST = render_GET def _delayedRender(self, request): - raw = getarg(request, 'raw', 'HTTP 1.1 200 OK\n') + raw = getarg(request, b'raw', b'HTTP 1.1 200 OK\n') request.startedWriting = 1 request.write(raw) request.channel.transport.loseConnection() @@ -125,29 +128,31 @@ class Echo(LeafResource): def render_GET(self, request): output = { - 'headers': dict(request.requestHeaders.getAllRawHeaders()), - 'body': request.content.read(), + 'headers': dict( + (to_unicode(k), [to_unicode(v) for v in vs]) + for k, vs in request.requestHeaders.getAllRawHeaders()), + 'body': to_unicode(request.content.read()), } - return json.dumps(output) + return to_bytes(json.dumps(output)) class Partial(LeafResource): def render_GET(self, request): - request.setHeader("Content-Length", "1024") + request.setHeader(b"Content-Length", b"1024") self.deferRequest(request, 0, self._delayedRender, request) return NOT_DONE_YET def _delayedRender(self, request): - request.write("partial content\n") + request.write(b"partial content\n") request.finish() class Drop(Partial): def _delayedRender(self, request): - abort = getarg(request, "abort", 0, type=int) - request.write("this connection will be dropped\n") + abort = getarg(request, b"abort", 0, type=int) + request.write(b"this connection will be dropped\n") tr = request.channel.transport try: if abort and hasattr(tr, 'abortConnection'): @@ -162,26 +167,26 @@ class Root(Resource): def __init__(self): Resource.__init__(self) - self.putChild("status", Status()) - self.putChild("follow", Follow()) - self.putChild("delay", Delay()) - self.putChild("partial", Partial()) - self.putChild("drop", Drop()) - self.putChild("raw", Raw()) - self.putChild("echo", Echo()) + self.putChild(b"status", Status()) + self.putChild(b"follow", Follow()) + self.putChild(b"delay", Delay()) + self.putChild(b"partial", Partial()) + self.putChild(b"drop", Drop()) + self.putChild(b"raw", Raw()) + self.putChild(b"echo", Echo()) - if six.PY2 and twisted_version > (12, 3, 0): + if twisted_version > (12, 3, 0): from twisted.web.test.test_webclient import PayloadResource from twisted.web.server import GzipEncoderFactory from twisted.web.resource import EncodingResourceWrapper - self.putChild('payload', PayloadResource()) - self.putChild("xpayload", EncodingResourceWrapper(PayloadResource(), [GzipEncoderFactory()])) + self.putChild(b"payload", PayloadResource()) + self.putChild(b"xpayload", EncodingResourceWrapper(PayloadResource(), [GzipEncoderFactory()])) def getChild(self, name, request): return self def render(self, request): - return 'Scrapy mock HTTP server\n' + return b'Scrapy mock HTTP server\n' class MockServer(): @@ -199,14 +204,18 @@ class MockServer(): time.sleep(0.2) +def ssl_context_factory(): + return ssl.DefaultOpenSSLContextFactory( + os.path.join(os.path.dirname(__file__), 'keys/cert.pem'), + os.path.join(os.path.dirname(__file__), 'keys/cert.pem'), + ) + + if __name__ == "__main__": root = Root() factory = Site(root) httpPort = reactor.listenTCP(8998, factory) - contextFactory = ssl.DefaultOpenSSLContextFactory( - os.path.join(os.path.dirname(__file__), 'keys/cert.pem'), - os.path.join(os.path.dirname(__file__), 'keys/cert.pem'), - ) + contextFactory = ssl_context_factory() httpsPort = reactor.listenSSL(8999, factory, contextFactory) def print_listening(): diff --git a/tests/py3-ignores.txt b/tests/py3-ignores.txt index 55ed75c92..f189a4c86 100644 --- a/tests/py3-ignores.txt +++ b/tests/py3-ignores.txt @@ -1,24 +1,10 @@ -tests/test_closespider.py -tests/test_command_fetch.py -tests/test_command_shell.py tests/test_exporters.py tests/test_linkextractors_deprecated.py -tests/test_crawl.py -tests/test_downloader_handlers.py -tests/test_downloadermiddleware_httpcache.py -tests/test_downloadermiddleware_httpcompression.py -tests/test_downloadermiddleware_httpproxy.py -tests/test_downloadermiddleware.py -tests/test_downloadermiddleware_retry.py -tests/test_engine.py tests/test_mail.py tests/test_pipeline_files.py tests/test_pipeline_images.py tests/test_proxy_connect.py tests/test_spidermiddleware_httperror.py -tests/test_utils_iterators.py -tests/test_utils_template.py -tests/test_webclient.py scrapy/xlib/tx/iweb.py scrapy/xlib/tx/interfaces.py @@ -27,18 +13,12 @@ scrapy/xlib/tx/client.py scrapy/xlib/tx/_newclient.py scrapy/xlib/tx/__init__.py scrapy/core/downloader/handlers/s3.py -scrapy/core/downloader/handlers/http11.py -scrapy/core/downloader/handlers/http.py scrapy/core/downloader/handlers/ftp.py -scrapy/core/downloader/webclient.py scrapy/pipelines/images.py scrapy/pipelines/files.py scrapy/linkextractors/sgml.py scrapy/linkextractors/regex.py scrapy/linkextractors/htmlparser.py -scrapy/downloadermiddlewares/retry.py -scrapy/downloadermiddlewares/httpcache.py -scrapy/downloadermiddlewares/httpproxy.py scrapy/downloadermiddlewares/cookies.py scrapy/extensions/statsmailer.py scrapy/extensions/memusage.py diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt index a709a734e..5cf786a89 100644 --- a/tests/requirements-py3.txt +++ b/tests/requirements-py3.txt @@ -3,6 +3,7 @@ pytest-twisted pytest-cov testfixtures jmespath +leveldb # optional for shell wrapper tests bpython ipython diff --git a/tests/spiders.py b/tests/spiders.py index 516062929..711d80cac 100644 --- a/tests/spiders.py +++ b/tests/spiders.py @@ -119,7 +119,7 @@ class BrokenStartRequestsSpider(FollowAllSpider): if self.fail_before_yield: 1 / 0 - for s in xrange(100): + for s in range(100): qargs = {'total': 10, 'seed': s} url = "http://localhost:8998/follow?%s" % urlencode(qargs, doseq=1) yield Request(url, meta={'seed': s}) diff --git a/tests/test_cmdline/__init__.py b/tests/test_cmdline/__init__.py index 5192fb0fa..c2de4fbc8 100644 --- a/tests/test_cmdline/__init__.py +++ b/tests/test_cmdline/__init__.py @@ -1,10 +1,11 @@ -import os import json -import sys -import shutil +import os import pstats -import tempfile +import shutil +import six from subprocess import Popen, PIPE +import sys +import tempfile import unittest try: from cStringIO import StringIO @@ -57,14 +58,14 @@ class CmdlineTest(unittest.TestCase): shutil.rmtree(path) def test_override_dict_settings(self): + EXT_PATH = "tests.test_cmdline.extensions.DummyExtension" + EXTENSIONS = {EXT_PATH: 200} settingsstr = self._execute('settings', '--get', 'EXTENSIONS', '-s', - ('EXTENSIONS={"tests.test_cmdline.extensions.TestExtension": ' - '100, "tests.test_cmdline.extensions.DummyExtension": 200}')) + 'EXTENSIONS=' + json.dumps(EXTENSIONS)) # XXX: There's gotta be a smarter way to do this... self.assertNotIn("...", settingsstr) for char in ("'", "<", ">", 'u"'): settingsstr = settingsstr.replace(char, '"') settingsdict = json.loads(settingsstr) - self.assertIn('tests.test_cmdline.extensions.DummyExtension', settingsdict) - self.assertIn('value=200', settingsdict['tests.test_cmdline.extensions.DummyExtension']) - self.assertIn('value=100', settingsdict['tests.test_cmdline.extensions.TestExtension']) + six.assertCountEqual(self, settingsdict.keys(), EXTENSIONS.keys()) + self.assertIn('value=200', settingsdict[EXT_PATH]) diff --git a/tests/test_command_fetch.py b/tests/test_command_fetch.py index 5283852b7..4843a9a2f 100644 --- a/tests/test_command_fetch.py +++ b/tests/test_command_fetch.py @@ -12,11 +12,11 @@ class FetchTest(ProcessTest, SiteTest, unittest.TestCase): @defer.inlineCallbacks def test_output(self): _, out, _ = yield self.execute([self.url('/text')]) - self.assertEqual(out.strip(), 'Works') + self.assertEqual(out.strip(), b'Works') @defer.inlineCallbacks def test_headers(self): _, out, _ = yield self.execute([self.url('/text'), '--headers']) - out = out.replace('\r', '') # required on win32 - assert 'Server: TwistedWeb' in out - assert 'Content-Type: text/plain' in out + out = out.replace(b'\r', b'') # required on win32 + assert b'Server: TwistedWeb' in out, out + assert b'Content-Type: text/plain' in out diff --git a/tests/test_command_shell.py b/tests/test_command_shell.py index a56236d54..105202754 100644 --- a/tests/test_command_shell.py +++ b/tests/test_command_shell.py @@ -12,38 +12,38 @@ class ShellTest(ProcessTest, SiteTest, unittest.TestCase): @defer.inlineCallbacks def test_empty(self): _, out, _ = yield self.execute(['-c', 'item']) - assert '{}' in out + assert b'{}' in out @defer.inlineCallbacks def test_response_body(self): _, out, _ = yield self.execute([self.url('/text'), '-c', 'response.body']) - assert 'Works' in out + assert b'Works' in out @defer.inlineCallbacks def test_response_type_text(self): _, out, _ = yield self.execute([self.url('/text'), '-c', 'type(response)']) - assert 'TextResponse' in out + assert b'TextResponse' in out @defer.inlineCallbacks def test_response_type_html(self): _, out, _ = yield self.execute([self.url('/html'), '-c', 'type(response)']) - assert 'HtmlResponse' in out + assert b'HtmlResponse' in out @defer.inlineCallbacks def test_response_selector_html(self): xpath = 'response.xpath("//p[@class=\'one\']/text()").extract()[0]' _, out, _ = yield self.execute([self.url('/html'), '-c', xpath]) - self.assertEqual(out.strip(), 'Works') + self.assertEqual(out.strip(), b'Works') @defer.inlineCallbacks def test_response_encoding_gb18030(self): _, out, _ = yield self.execute([self.url('/enc-gb18030'), '-c', 'response.encoding']) - self.assertEqual(out.strip(), 'gb18030') + self.assertEqual(out.strip(), b'gb18030') @defer.inlineCallbacks def test_redirect(self): _, out, _ = yield self.execute([self.url('/redirect'), '-c', 'response.url']) - assert out.strip().endswith('/redirected') + assert out.strip().endswith(b'/redirected') @defer.inlineCallbacks def test_request_replace(self): diff --git a/tests/test_commands.py b/tests/test_commands.py index e0c0648ca..a5230eb13 100644 --- a/tests/test_commands.py +++ b/tests/test_commands.py @@ -4,13 +4,14 @@ import subprocess import tempfile from time import sleep from os.path import exists, join, abspath -from shutil import rmtree +from shutil import rmtree, copytree from tempfile import mkdtemp import six from twisted.trial import unittest from twisted.internet import defer +import scrapy from scrapy.utils.python import to_native_str from scrapy.utils.python import retry_on_eintr from scrapy.utils.test import get_testenv @@ -73,6 +74,27 @@ class StartprojectTest(ProjectTest): self.assertEqual(1, self.call('startproject', 'sys')) +class StartprojectTemplatesTest(ProjectTest): + + def setUp(self): + super(StartprojectTemplatesTest, self).setUp() + self.tmpl = join(self.temp_path, 'templates') + self.tmpl_proj = join(self.tmpl, 'project') + + def test_startproject_template_override(self): + copytree(join(scrapy.__path__[0], 'templates'), self.tmpl) + with open(join(self.tmpl_proj, 'root_template'), 'w'): + pass + assert exists(join(self.tmpl_proj, 'root_template')) + + args = ['--set', 'TEMPLATES_DIR=%s' % self.tmpl] + p = self.proc('startproject', self.project_name, *args) + out = to_native_str(retry_on_eintr(p.stdout.read)) + self.assertIn("New Scrapy project %r, using template directory" % self.project_name, out) + self.assertIn(self.tmpl_proj, out) + assert exists(join(self.proj_path, 'root_template')) + + class CommandTest(ProjectTest): def setUp(self): @@ -255,3 +277,4 @@ class BenchCommandTest(CommandTest): '-s', 'CLOSESPIDER_TIMEOUT=0.01') log = to_native_str(p.stderr.read()) self.assertIn('INFO: Crawled', log) + self.assertNotIn('Unhandled Error', log) diff --git a/tests/test_crawl.py b/tests/test_crawl.py index 814eb30d2..90fd921c8 100644 --- a/tests/test_crawl.py +++ b/tests/test_crawl.py @@ -8,6 +8,7 @@ from twisted.trial.unittest import TestCase from scrapy.http import Request from scrapy.crawler import CrawlerRunner +from scrapy.utils.python import to_unicode from tests import mock from tests.spiders import FollowAllSpider, DelaySpider, SimpleSpider, \ BrokenStartRequestsSpider, SingleRequestSpider, DuplicateStartRequestsSpider @@ -201,16 +202,16 @@ with multiples lines self.assertIn('responses', crawler.spider.meta) self.assertNotIn('failures', crawler.spider.meta) # start requests doesn't set Referer header - echo0 = json.loads(crawler.spider.meta['responses'][2].body) + echo0 = json.loads(to_unicode(crawler.spider.meta['responses'][2].body)) self.assertNotIn('Referer', echo0['headers']) # following request sets Referer to start request url - echo1 = json.loads(crawler.spider.meta['responses'][1].body) + echo1 = json.loads(to_unicode(crawler.spider.meta['responses'][1].body)) self.assertEqual(echo1['headers'].get('Referer'), [req0.url]) # next request avoids Referer header - echo2 = json.loads(crawler.spider.meta['responses'][2].body) + echo2 = json.loads(to_unicode(crawler.spider.meta['responses'][2].body)) self.assertNotIn('Referer', echo2['headers']) # last request explicitly sets a Referer header - echo3 = json.loads(crawler.spider.meta['responses'][3].body) + echo3 = json.loads(to_unicode(crawler.spider.meta['responses'][3].body)) self.assertEqual(echo3['headers'].get('Referer'), ['http://example.com']) @defer.inlineCallbacks diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index e4d957d8e..1eb6192ce 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -1,5 +1,4 @@ import os -import twisted import six from twisted.trial import unittest @@ -10,9 +9,7 @@ from twisted.web import server, static, util, resource from twisted.web.test.test_webclient import ForeverTakingResource, \ NoLengthResource, HostHeaderResource, \ PayloadResource, BrokenDownloadResource -from twisted.protocols.ftp import FTPRealm, FTPFactory from twisted.cred import portal, checkers, credentials -from twisted.protocols.ftp import FTPClient, ConnectionLost from w3lib.url import path_to_file_uri from scrapy import twisted_version @@ -22,16 +19,15 @@ from scrapy.core.downloader.handlers.http import HTTPDownloadHandler, HttpDownlo from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler from scrapy.core.downloader.handlers.s3 import S3DownloadHandler -from scrapy.core.downloader.handlers.ftp import FTPDownloadHandler from scrapy.spiders import Spider from scrapy.http import Request from scrapy.settings import Settings -from scrapy import optional_features from scrapy.utils.test import get_crawler +from scrapy.utils.python import to_bytes from scrapy.exceptions import NotConfigured -from tests.mockserver import MockServer +from tests.mockserver import MockServer, ssl_context_factory from tests.spiders import SingleRequestSpider class DummyDH(object): @@ -92,7 +88,7 @@ class FileTestCase(unittest.TestCase): def _test(response): self.assertEquals(response.url, request.url) self.assertEquals(response.status, 200) - self.assertEquals(response.body, '0123456789') + self.assertEquals(response.body, b'0123456789') request = Request(path_to_file_uri(self.tmpname + '^')) assert request.url.upper().endswith('%5E') @@ -106,23 +102,29 @@ class FileTestCase(unittest.TestCase): class HttpTestCase(unittest.TestCase): + scheme = 'http' download_handler_cls = HTTPDownloadHandler def setUp(self): name = self.mktemp() os.mkdir(name) - FilePath(name).child("file").setContent("0123456789") + FilePath(name).child("file").setContent(b"0123456789") r = static.File(name) - r.putChild("redirect", util.Redirect("/file")) - r.putChild("wait", ForeverTakingResource()) - r.putChild("hang-after-headers", ForeverTakingResource(write=True)) - r.putChild("nolength", NoLengthResource()) - r.putChild("host", HostHeaderResource()) - r.putChild("payload", PayloadResource()) - r.putChild("broken", BrokenDownloadResource()) + r.putChild(b"redirect", util.Redirect(b"/file")) + r.putChild(b"wait", ForeverTakingResource()) + r.putChild(b"hang-after-headers", ForeverTakingResource(write=True)) + r.putChild(b"nolength", NoLengthResource()) + r.putChild(b"host", HostHeaderResource()) + r.putChild(b"payload", PayloadResource()) + r.putChild(b"broken", BrokenDownloadResource()) self.site = server.Site(r, timeout=None) self.wrapper = WrappingFactory(self.site) - self.port = reactor.listenTCP(0, self.wrapper, interface='127.0.0.1') + self.host = 'localhost' + if self.scheme == 'https': + self.port = reactor.listenSSL( + 0, self.wrapper, ssl_context_factory(), interface=self.host) + else: + self.port = reactor.listenTCP(0, self.wrapper, interface=self.host) self.portno = self.port.getHost().port self.download_handler = self.download_handler_cls(Settings()) self.download_request = self.download_handler.download_request @@ -134,20 +136,20 @@ class HttpTestCase(unittest.TestCase): yield self.download_handler.close() def getURL(self, path): - return "http://127.0.0.1:%d/%s" % (self.portno, path) + return "%s://%s:%d/%s" % (self.scheme, self.host, self.portno, path) def test_download(self): request = Request(self.getURL('file')) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) - d.addCallback(self.assertEquals, "0123456789") + d.addCallback(self.assertEquals, b"0123456789") return d def test_download_head(self): request = Request(self.getURL('file'), method='HEAD') d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) - d.addCallback(self.assertEquals, '') + d.addCallback(self.assertEquals, b'') return d def test_redirect_status(self): @@ -166,6 +168,9 @@ class HttpTestCase(unittest.TestCase): @defer.inlineCallbacks def test_timeout_download_from_spider(self): + if self.scheme == 'https': + raise unittest.SkipTest( + 'test_timeout_download_from_spider skipped under https') spider = Spider('foo') meta = {'download_timeout': 0.2} # client connects but no data is received @@ -179,7 +184,8 @@ class HttpTestCase(unittest.TestCase): def test_host_header_not_in_request_headers(self): def _test(response): - self.assertEquals(response.body, '127.0.0.1:%d' % self.portno) + self.assertEquals( + response.body, to_bytes('%s:%d' % (self.host, self.portno))) self.assertEquals(request.headers, {}) request = Request(self.getURL('host')) @@ -187,19 +193,19 @@ class HttpTestCase(unittest.TestCase): def test_host_header_seted_in_request_headers(self): def _test(response): - self.assertEquals(response.body, 'example.com') - self.assertEquals(request.headers.get('Host'), 'example.com') + self.assertEquals(response.body, b'example.com') + self.assertEquals(request.headers.get('Host'), b'example.com') request = Request(self.getURL('host'), headers={'Host': 'example.com'}) return self.download_request(request, Spider('foo')).addCallback(_test) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) - d.addCallback(self.assertEquals, 'example.com') + d.addCallback(self.assertEquals, b'example.com') return d def test_payload(self): - body = '1'*100 # PayloadResource requires body length to be 100 + body = b'1'*100 # PayloadResource requires body length to be 100 request = Request(self.getURL('payload'), method='POST', body=body) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) @@ -217,17 +223,21 @@ class Http10TestCase(HttpTestCase): download_handler_cls = HTTP10DownloadHandler +class Https10TestCase(Http10TestCase): + scheme = 'https' + + class Http11TestCase(HttpTestCase): """HTTP 1.1 test case""" download_handler_cls = HTTP11DownloadHandler - if 'http11' not in optional_features: + if twisted_version < (11, 1, 0): skip = 'HTTP1.1 not supported in twisted < 11.1.0' def test_download_without_maxsize_limit(self): request = Request(self.getURL('file')) d = self.download_request(request, Spider('foo')) d.addCallback(lambda r: r.body) - d.addCallback(self.assertEquals, "0123456789") + d.addCallback(self.assertEquals, b"0123456789") return d @defer.inlineCallbacks @@ -238,7 +248,7 @@ class Http11TestCase(HttpTestCase): # response body. (regardless of headers) d = self.download_request(request, Spider('foo', download_maxsize=10)) d.addCallback(lambda r: r.body) - d.addCallback(self.assertEquals, "0123456789") + d.addCallback(self.assertEquals, b"0123456789") yield d d = self.download_request(request, Spider('foo', download_maxsize=9)) @@ -261,13 +271,17 @@ class Http11TestCase(HttpTestCase): request = Request(self.getURL('file')) d = self.download_request(request, Spider('foo', download_maxsize=100)) d.addCallback(lambda r: r.body) - d.addCallback(self.assertEquals, "0123456789") + d.addCallback(self.assertEquals, b"0123456789") return d +class Https11TestCase(Http11TestCase): + scheme = 'https' + + class Http11MockServerTestCase(unittest.TestCase): """HTTP 1.1 test case with MockServer""" - if 'http11' not in optional_features: + if twisted_version < (11, 1, 0): skip = 'HTTP1.1 not supported in twisted < 11.1.0' def setUp(self): @@ -298,27 +312,30 @@ class Http11MockServerTestCase(unittest.TestCase): @defer.inlineCallbacks def test_download_gzip_response(self): - if six.PY2 and twisted_version > (12, 3, 0): + if twisted_version > (12, 3, 0): crawler = get_crawler(SingleRequestSpider) - body = '1'*100 # PayloadResource requires body length to be 100 + body = b'1'*100 # PayloadResource requires body length to be 100 request = Request('http://localhost:8998/payload', method='POST', body=body, meta={'download_maxsize': 50}) yield crawler.crawl(seed=request) failure = crawler.spider.meta['failure'] # download_maxsize < 100, hence the CancelledError self.assertIsInstance(failure.value, defer.CancelledError) - request.headers.setdefault('Accept-Encoding', 'gzip,deflate') - request = request.replace(url='http://localhost:8998/xpayload') - yield crawler.crawl(seed=request) - - # download_maxsize = 50 is enough for the gzipped response - failure = crawler.spider.meta.get('failure') - self.assertTrue(failure == None) - reason = crawler.spider.meta['close_reason'] - self.assertTrue(reason, 'finished') + if six.PY2: + request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate') + request = request.replace(url='http://localhost:8998/xpayload') + yield crawler.crawl(seed=request) + # download_maxsize = 50 is enough for the gzipped response + failure = crawler.spider.meta.get('failure') + self.assertTrue(failure == None) + reason = crawler.spider.meta['close_reason'] + self.assertTrue(reason, 'finished') + else: + # See issue https://twistedmatrix.com/trac/ticket/8175 + raise unittest.SkipTest("xpayload only enabled for PY2") else: - raise unittest.SkipTest("xpayload and payload endpoint only enabled for twisted > 12.3.0 and python 2.x") + raise unittest.SkipTest("xpayload and payload endpoint only enabled for twisted > 12.3.0") class UriResource(resource.Resource): @@ -355,7 +372,7 @@ class HttpProxyTestCase(unittest.TestCase): def _test(response): self.assertEquals(response.status, 200) self.assertEquals(response.url, request.url) - self.assertEquals(response.body, 'http://example.com') + self.assertEquals(response.body, b'http://example.com') http_proxy = self.getURL('') request = Request('http://example.com', meta={'proxy': http_proxy}) @@ -365,7 +382,7 @@ class HttpProxyTestCase(unittest.TestCase): def _test(response): self.assertEquals(response.status, 200) self.assertEquals(response.url, request.url) - self.assertEquals(response.body, 'https://example.com') + self.assertEquals(response.body, b'https://example.com') http_proxy = '%s?noconnect' % self.getURL('') request = Request('https://example.com', meta={'proxy': http_proxy}) @@ -375,7 +392,7 @@ class HttpProxyTestCase(unittest.TestCase): def _test(response): self.assertEquals(response.status, 200) self.assertEquals(response.url, request.url) - self.assertEquals(response.body, '/path/to/resource') + self.assertEquals(response.body, b'/path/to/resource') request = Request(self.getURL('path/to/resource')) return self.download_request(request, Spider('foo')).addCallback(_test) @@ -392,9 +409,20 @@ class Http10ProxyTestCase(HttpProxyTestCase): class Http11ProxyTestCase(HttpProxyTestCase): download_handler_cls = HTTP11DownloadHandler - if 'http11' not in optional_features: + if twisted_version < (11, 1, 0): skip = 'HTTP1.1 not supported in twisted < 11.1.0' + @defer.inlineCallbacks + def test_download_with_proxy_https_timeout(self): + """ Test TunnelingTCP4ClientEndpoint """ + http_proxy = self.getURL('') + domain = 'https://no-such-domain.nosuch' + request = Request( + domain, meta={'proxy': http_proxy, 'download_timeout': 0.2}) + d = self.download_request(request, Spider('foo')) + timeout = yield self.assertFailure(d, error.TimeoutError) + self.assertIn(domain, timeout.osError) + class HttpDownloadHandlerMock(object): def __init__(self, settings): @@ -403,14 +431,34 @@ class HttpDownloadHandlerMock(object): def download_request(self, request, spider): return request + +class S3AnonTestCase(unittest.TestCase): + try: + import boto + except ImportError: + skip = 'missing boto library' + + def setUp(self): + self.s3reqh = S3DownloadHandler(Settings(), + httpdownloadhandler=HttpDownloadHandlerMock, + #anon=True, # is implicit + ) + self.download_request = self.s3reqh.download_request + self.spider = Spider('foo') + + def test_anon_request(self): + req = Request('s3://aws-publicdatasets/') + httpreq = self.download_request(req, self.spider) + self.assertEqual(hasattr(self.s3reqh.conn, 'anon'), True) + self.assertEqual(self.s3reqh.conn.anon, True) + + class S3TestCase(unittest.TestCase): download_handler_cls = S3DownloadHandler try: - # can't instance without settings, but ignore that - download_handler_cls({}) - except NotConfigured: + import boto + except ImportError: skip = 'missing boto library' - except KeyError: pass # test use same example keys than amazon developer guide # http://s3.amazonaws.com/awsdocs/S3/20060301/s3-dg-20060301.pdf @@ -420,8 +468,8 @@ class S3TestCase(unittest.TestCase): AWS_SECRET_ACCESS_KEY = 'uV3F3YluFJax1cknvbcGwgjvx4QpvB+leU8dUj2o' def setUp(self): - s3reqh = S3DownloadHandler(Settings(), self.AWS_ACCESS_KEY_ID, \ - self.AWS_SECRET_ACCESS_KEY, \ + s3reqh = S3DownloadHandler(Settings(), self.AWS_ACCESS_KEY_ID, + self.AWS_SECRET_ACCESS_KEY, httpdownloadhandler=HttpDownloadHandlerMock) self.download_request = s3reqh.download_request self.spider = Spider('foo') @@ -519,8 +567,13 @@ class FTPTestCase(unittest.TestCase): if twisted_version < (10, 2, 0): skip = "Twisted pre 10.2.0 doesn't allow to set home path other than /home" + if six.PY3: + skip = "Twisted missing ftp support for PY3" def setUp(self): + from twisted.protocols.ftp import FTPRealm, FTPFactory + from scrapy.core.downloader.handlers.ftp import FTPDownloadHandler + # setup dirs and test file self.directory = self.mktemp() os.mkdir(self.directory) @@ -602,6 +655,8 @@ class FTPTestCase(unittest.TestCase): return self._add_test_callbacks(d, _test) def test_invalid_credentials(self): + from twisted.protocols.ftp import ConnectionLost + request = Request(url="ftp://127.0.0.1:%s/file.txt" % self.portNum, meta={"ftp_user": self.username, "ftp_password": 'invalid'}) d = self.download_handler.download_request(request, None) diff --git a/tests/test_downloadermiddleware.py b/tests/test_downloadermiddleware.py index 13f35b92a..fb51392b2 100644 --- a/tests/test_downloadermiddleware.py +++ b/tests/test_downloadermiddleware.py @@ -5,6 +5,7 @@ from scrapy.http import Request, Response from scrapy.spiders import Spider from scrapy.core.downloader.middleware import DownloaderMiddlewareManager from scrapy.utils.test import get_crawler +from scrapy.utils.python import to_bytes from tests import mock @@ -68,7 +69,7 @@ class DefaultsTest(ManagerTestCase): """ req = Request('http://example.com') - body = '

You are being redirected

' + body = b'

You are being redirected

' resp = Response(req.url, status=302, body=body, headers={ 'Content-Length': str(len(body)), 'Content-Type': 'text/html', @@ -78,12 +79,12 @@ class DefaultsTest(ManagerTestCase): ret = self._download(request=req, response=resp) self.assertTrue(isinstance(ret, Request), "Not redirected: {0!r}".format(ret)) - self.assertEqual(ret.url, resp.headers['Location'], + self.assertEqual(to_bytes(ret.url), resp.headers['Location'], "Not redirected to location header") def test_200_and_invalid_gzipped_body_must_fail(self): req = Request('http://example.com') - body = '

You are being redirected

' + body = b'

You are being redirected

' resp = Response(req.url, status=200, body=body, headers={ 'Content-Length': str(len(body)), 'Content-Type': 'text/html', diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 53389ae3b..12b69860a 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -31,7 +31,7 @@ class _BaseTest(unittest.TestCase): headers={'User-Agent': 'test'}) self.response = Response('http://www.example.com', headers={'Content-Type': 'text/html'}, - body='test body', + body=b'test body', status=202) self.crawler.stats.open_spider(self.spider) @@ -84,9 +84,9 @@ class _BaseTest(unittest.TestCase): def assertEqualRequestButWithCacheValidators(self, request1, request2): self.assertEqual(request1.url, request2.url) - assert not 'If-None-Match' in request1.headers - assert not 'If-Modified-Since' in request1.headers - assert any(h in request2.headers for h in ('If-None-Match', 'If-Modified-Since')) + assert not b'If-None-Match' in request1.headers + assert not b'If-Modified-Since' in request1.headers + assert any(h in request2.headers for h in (b'If-None-Match', b'If-Modified-Since')) self.assertEqual(request1.body, request2.body) def test_dont_cache(self): @@ -257,6 +257,7 @@ class RFC2616PolicyTest(DefaultStorageTest): policy_class = 'scrapy.extensions.httpcache.RFC2616Policy' def _process_requestresponse(self, mw, request, response): + result = None try: result = mw.process_request(request, self.spider) if result: @@ -291,7 +292,7 @@ class RFC2616PolicyTest(DefaultStorageTest): self.assertEqualResponse(res2, res3) # request with no-cache directive must not return cached response # but it allows new response to be stored - res0b = res0.replace(body='foo') + res0b = res0.replace(body=b'foo') res4 = self._process_requestresponse(mw, req2, res0b) self.assertEqualResponse(res4, res0b) assert 'cached' not in res4.flags @@ -435,7 +436,7 @@ class RFC2616PolicyTest(DefaultStorageTest): assert 'cached' not in res1.flags # Same request but as cached response is stale a new response must # be returned - res0b = res0a.replace(body='bar') + res0b = res0a.replace(body=b'bar') res2 = self._process_requestresponse(mw, req0, res0b) self.assertEqualResponse(res2, res0b) assert 'cached' not in res2.flags diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index a18994ef3..2e6e47fef 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -50,46 +50,46 @@ class HttpCompressionTest(TestCase): request = Request('http://scrapytest.org') assert 'Accept-Encoding' not in request.headers self.mw.process_request(request, self.spider) - self.assertEqual(request.headers.get('Accept-Encoding'), 'gzip,deflate') + self.assertEqual(request.headers.get('Accept-Encoding'), b'gzip,deflate') def test_process_response_gzip(self): response = self._getresponse('gzip') request = response.request - self.assertEqual(response.headers['Content-Encoding'], 'gzip') + self.assertEqual(response.headers['Content-Encoding'], b'gzip') newresponse = self.mw.process_response(request, response, self.spider) assert newresponse is not response - assert newresponse.body.startswith('= (11, 1, 0): # http11 available exceptions.append(ResponseFailed) for exc in exceptions: diff --git a/tests/test_engine.py b/tests/test_engine.py index dad921a60..9f2c02bff 100644 --- a/tests/test_engine.py +++ b/tests/test_engine.py @@ -55,11 +55,12 @@ class TestSpider(Spider): def parse_item(self, response): item = self.item_cls() - m = self.name_re.search(response.body) + body = response.body_as_unicode() + m = self.name_re.search(body) if m: item['name'] = m.group(1) item['url'] = response.url - m = self.price_re.search(response.body) + m = self.price_re.search(body) if m: item['price'] = m.group(1) return item @@ -77,8 +78,8 @@ class DictItemsSpider(TestSpider): def start_test_site(debug=False): root_dir = os.path.join(tests_datadir, "test_site") r = static.File(root_dir) - r.putChild("redirect", util.Redirect("/redirected")) - r.putChild("redirected", static.Data("Redirected here", "text/plain")) + r.putChild(b"redirect", util.Redirect(b"/redirected")) + r.putChild(b"redirected", static.Data(b"Redirected here", "text/plain")) port = reactor.listenTCP(0, server.Site(r), interface="127.0.0.1") if debug: @@ -237,12 +238,12 @@ class EngineTest(unittest.TestCase): @defer.inlineCallbacks def test_close_downloader(self): - e = ExecutionEngine(get_crawler(TestSpider), lambda: None) + e = ExecutionEngine(get_crawler(TestSpider), lambda _: None) yield e.close() @defer.inlineCallbacks def test_close_spiders_downloader(self): - e = ExecutionEngine(get_crawler(TestSpider), lambda: None) + e = ExecutionEngine(get_crawler(TestSpider), lambda _: None) yield e.open_spider(TestSpider(), []) self.assertEqual(len(e.open_spiders), 1) yield e.close() @@ -250,7 +251,7 @@ class EngineTest(unittest.TestCase): @defer.inlineCallbacks def test_close_engine_spiders_downloader(self): - e = ExecutionEngine(get_crawler(TestSpider), lambda: None) + e = ExecutionEngine(get_crawler(TestSpider), lambda _: None) yield e.open_spider(TestSpider(), []) e.start() self.assertTrue(e.running) diff --git a/tests/test_http_request.py b/tests/test_http_request.py index ff0941961..b81d43c41 100644 --- a/tests/test_http_request.py +++ b/tests/test_http_request.py @@ -1,5 +1,6 @@ import cgi import unittest +import re import six from six.moves import xmlrpc_client as xmlrpclib @@ -305,6 +306,19 @@ class FormRequestTest(RequestTest): request = FormRequest.from_response(response, url='/relative') self.assertEqual(request.url, 'http://example.com/relative') + def test_from_response_case_insensitive(self): + response = _buildresponse( + """ + + + + """) + req = self.request_class.from_response(response) + fs = _qs(req) + self.assertEqual(fs['clickable1'], ['clicked1']) + self.assertFalse('i1' in fs, fs) # xpath in _get_inputs() + self.assertFalse('clickable2' in fs, fs) # xpath in _get_clickable() + def test_from_response_submit_first_clickable(self): response = _buildresponse( """
@@ -662,10 +676,11 @@ class FormRequestTest(RequestTest): +
''') req = self.request_class.from_response(res) fs = _qs(req) - self.assertEqual(fs, {'i1': ['i1v1'], 'i2': ['']}) + self.assertEqual(fs, {'i1': ['i1v1'], 'i2': [''], 'i4': ['i4v1']}) def test_from_response_input_hidden(self): res = _buildresponse( @@ -733,6 +748,18 @@ class FormRequestTest(RequestTest): self.assertRaises(ValueError, self.request_class.from_response, response, formxpath="//form/input[@name='abc']") + def test_from_response_unicode_xpath(self): + response = _buildresponse(b'
') + r = self.request_class.from_response(response, formxpath=u"//form[@name='\u044a']") + fs = _qs(r) + self.assertEqual(fs, {}) + + xpath = u"//form[@name='\u03b1']" + encoded = xpath if six.PY3 else xpath.encode('unicode_escape') + self.assertRaisesRegexp(ValueError, re.escape(encoded), + self.request_class.from_response, + response, formxpath=xpath) + def test_from_response_button_submit(self): response = _buildresponse( """
@@ -801,6 +828,45 @@ class FormRequestTest(RequestTest): self.assertEqual(fs[b'test2'], [b'val2']) self.assertEqual(fs[b'button1'], [b'']) + def test_html_base_form_action(self): + response = _buildresponse( + """ + + + + + + +
+ + + """, + url='http://a.com/' + ) + req = self.request_class.from_response(response) + self.assertEqual(req.url, 'http://b.com/test_form') + + def test_from_response_css(self): + response = _buildresponse( + """
+ + +
+
+ + +
""") + r1 = self.request_class.from_response(response, formcss="form[action='post.php']") + fs = _qs(r1) + self.assertEqual(fs[b'one'], [b'1']) + + r1 = self.request_class.from_response(response, formcss="input[name='four']") + fs = _qs(r1) + self.assertEqual(fs[b'three'], [b'3']) + + self.assertRaises(ValueError, self.request_class.from_response, + response, formcss="input[name='abc']") + def _buildresponse(body, **kwargs): kwargs.setdefault('body', body) kwargs.setdefault('url', 'http://example.com') diff --git a/tests/test_http_response.py b/tests/test_http_response.py index b49d46ea8..710a5b29d 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -17,6 +17,8 @@ class BaseResponseTest(unittest.TestCase): # Response requires url in the consturctor self.assertRaises(Exception, self.response_class) self.assertTrue(isinstance(self.response_class('http://example.com/'), self.response_class)) + if not six.PY2: + self.assertRaises(TypeError, self.response_class, b"http://example.com") # body can be str or None self.assertTrue(isinstance(self.response_class('http://example.com/', body=b''), self.response_class)) self.assertTrue(isinstance(self.response_class('http://example.com/', body=b'body'), self.response_class)) diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py index e3664f8d8..89dcb75c2 100644 --- a/tests/test_linkextractors_deprecated.py +++ b/tests/test_linkextractors_deprecated.py @@ -190,3 +190,20 @@ class RegexLinkExtractorTestCase(unittest.TestCase): Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False), Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False), ]) + + def test_html_base_href(self): + html = """ + + + + + + + + + """ + response = HtmlResponse("http://a.com/", body=html) + lx = RegexLinkExtractor() + self.assertEqual([link for link in lx.extract_links(response)], [ + Link(url='http://b.com/test.html', text=u'', nofollow=False), + ]) diff --git a/tests/test_settings/__init__.py b/tests/test_settings/__init__.py index 4ef08bb0b..44b9b6df3 100644 --- a/tests/test_settings/__init__.py +++ b/tests/test_settings/__init__.py @@ -37,21 +37,22 @@ class SettingsAttributeTest(unittest.TestCase): self.assertEqual(self.attribute.value, 'value') self.assertEqual(self.attribute.priority, 10) - def test_set_per_key_priorities(self): - attribute = SettingsAttribute( - BaseSettings({'one': 10, 'two': 20}, 0), 0) + def test_overwrite_basesettings(self): + original_dict = {'one': 10, 'two': 20} + original_settings = BaseSettings(original_dict, 0) + attribute = SettingsAttribute(original_settings, 0) - new_dict = {'one': 11, 'two': 21} + new_dict = {'three': 11, 'four': 21} attribute.set(new_dict, 10) - self.assertEqual(attribute.value['one'], 11) - self.assertEqual(attribute.value['two'], 21) + self.assertIsInstance(attribute.value, BaseSettings) + six.assertCountEqual(self, attribute.value, new_dict) + six.assertCountEqual(self, original_settings, original_dict) - new_settings = BaseSettings() - new_settings.set('one', 12, 20) - new_settings.set('two', 12, 0) - attribute.set(new_settings, 0) - self.assertEqual(attribute.value['one'], 12) - self.assertEqual(attribute.value['two'], 21) + new_settings = BaseSettings({'five': 12}, 0) + attribute.set(new_settings, 0) # Insufficient priority + six.assertCountEqual(self, attribute.value, new_dict) + attribute.set(new_settings, 10) + six.assertCountEqual(self, attribute.value, new_settings) def test_repr(self): self.assertEqual(repr(self.attribute), @@ -263,24 +264,15 @@ class BaseSettingsTest(unittest.TestCase): self.assertEqual(settings.getpriority('key'), 99) self.assertEqual(settings.getpriority('nonexistentkey'), None) - def test_getcomposite(self): - s = BaseSettings({'TEST_BASE': {1: 1, 2: 2}, + def test_getwithbase(self): + s = BaseSettings({'TEST_BASE': BaseSettings({1: 1, 2: 2}, 'project'), 'TEST': BaseSettings({1: 10, 3: 30}, 'default'), - 'HASNOBASE': BaseSettings({1: 1}, 'default')}) - s['TEST'].set(4, 4, priority='project') - # When users specify a _BASE setting they explicitly don't want to use - # Scrapy's defaults, so we don't want to see anything that has a - # 'default' priority from TEST - cs = s._getcomposite('TEST') - self.assertEqual(len(cs), 3) - self.assertEqual(cs[1], 1) - self.assertEqual(cs[2], 2) - self.assertEqual(cs[4], 4) - cs = s._getcomposite('HASNOBASE') - self.assertEqual(len(cs), 1) - self.assertEqual(cs[1], 1) - cs = s._getcomposite('NONEXISTENT') - self.assertIsNone(cs) + 'HASNOBASE': BaseSettings({3: 3000}, 'default')}) + s['TEST'].set(2, 200, 'cmdline') + six.assertCountEqual(self, s.getwithbase('TEST'), + {1: 1, 2: 200, 3: 30}) + six.assertCountEqual(self, s.getwithbase('HASNOBASE'), s['HASNOBASE']) + self.assertEqual(s.getwithbase('NONEXISTENT'), {}) def test_maxpriority(self): # Empty settings should return 'default' diff --git a/tests/test_toplevel.py b/tests/test_toplevel.py index e9f220092..91bbe43bc 100644 --- a/tests/test_toplevel.py +++ b/tests/test_toplevel.py @@ -11,10 +11,6 @@ class ToplevelTestCase(TestCase): def test_version_info(self): self.assertIs(type(scrapy.version_info), tuple) - def test_optional_features(self): - self.assertIs(type(scrapy.optional_features), set) - self.assertIn('ssl', scrapy.optional_features) - def test_request_shortcut(self): from scrapy.http import Request, FormRequest self.assertIs(scrapy.Request, Request) diff --git a/tests/test_utils_conf.py b/tests/test_utils_conf.py index af15d3184..dab41ac8d 100644 --- a/tests/test_utils_conf.py +++ b/tests/test_utils_conf.py @@ -8,46 +8,59 @@ class BuildComponentListTest(unittest.TestCase): def test_build_dict(self): d = {'one': 1, 'two': None, 'three': 8, 'four': 4} - self.assertEqual(build_component_list(d, lambda x: x), + self.assertEqual(build_component_list(d, convert=lambda x: x), ['one', 'four', 'three']) + def test_backwards_compatible_build_dict(self): + base = {'one': 1, 'two': 2, 'three': 3, 'five': 5, 'six': None} + custom = {'two': None, 'three': 8, 'four': 4} + self.assertEqual(build_component_list(base, custom, + convert=lambda x: x), + ['one', 'four', 'five', 'three']) + def test_return_list(self): custom = ['a', 'b', 'c'] - self.assertEqual(build_component_list(custom, lambda x: x), custom) + self.assertEqual(build_component_list(None, custom, + convert=lambda x: x), + custom) def test_map_dict(self): custom = {'one': 1, 'two': 2, 'three': 3} - self.assertEqual(build_component_list(custom, lambda x: x.upper()), + self.assertEqual(build_component_list({}, custom, + convert=lambda x: x.upper()), ['ONE', 'TWO', 'THREE']) def test_map_list(self): custom = ['a', 'b', 'c'] - self.assertEqual(build_component_list(custom, lambda x: x.upper()), + self.assertEqual(build_component_list(None, custom, + lambda x: x.upper()), ['A', 'B', 'C']) def test_duplicate_components_in_dict(self): duplicate_dict = {'one': 1, 'two': 2, 'ONE': 4} - self.assertRaises(ValueError, - build_component_list, duplicate_dict, lambda x: x.lower()) + self.assertRaises(ValueError, build_component_list, {}, duplicate_dict, + convert=lambda x: x.lower()) def test_duplicate_components_in_list(self): duplicate_list = ['a', 'b', 'a'] - self.assertRaises(ValueError, - build_component_list, duplicate_list, lambda x: x) + self.assertRaises(ValueError, build_component_list, None, + duplicate_list, convert=lambda x: x) def test_duplicate_components_in_basesettings(self): # Higher priority takes precedence duplicate_bs = BaseSettings({'one': 1, 'two': 2}, priority=0) duplicate_bs.set('ONE', 4, priority=10) - self.assertEqual(build_component_list(duplicate_bs, convert=lambda x: x.lower()), + self.assertEqual(build_component_list(duplicate_bs, + convert=lambda x: x.lower()), ['two', 'one']) duplicate_bs.set('one', duplicate_bs['one'], priority=20) - self.assertEqual(build_component_list(duplicate_bs, convert=lambda x: x.lower()), + self.assertEqual(build_component_list(duplicate_bs, + convert=lambda x: x.lower()), ['one', 'two']) # Same priority raises ValueError duplicate_bs.set('ONE', duplicate_bs['ONE'], priority=20) - self.assertRaises(ValueError, - build_component_list, duplicate_bs, convert=lambda x: x.lower()) + self.assertRaises(ValueError, build_component_list, duplicate_bs, + convert=lambda x: x.lower()) class UtilsConfTestCase(unittest.TestCase): diff --git a/tests/test_utils_iterators.py b/tests/test_utils_iterators.py index 590c53302..b2e3889a4 100644 --- a/tests/test_utils_iterators.py +++ b/tests/test_utils_iterators.py @@ -1,4 +1,6 @@ +# -*- coding: utf-8 -*- import os +import six from twisted.trial import unittest from scrapy.utils.iterators import csviter, xmliter, _body_or_str, xmliter_lxml @@ -45,6 +47,60 @@ class XmliterTestCase(unittest.TestCase): for e in self.xmliter(response, 'matchme...')] self.assertEqual(nodenames, [['matchme...']]) + def test_xmliter_unicode(self): + # example taken from https://github.com/scrapy/scrapy/issues/1665 + body = u""" + <þingflokkar> + <þingflokkur id="26"> + + + - + + + + 80 + + + <þingflokkur id="21"> + Alþýðubandalag + + Ab + Alþb. + + + 76 + 123 + + + <þingflokkur id="27"> + Alþýðuflokkur + + A + Alþfl. + + + 27 + 120 + + + """ + + for r in ( + # with bytes + XmlResponse(url="http://example.com", body=body.encode('utf-8')), + # Unicode body needs encoding information + XmlResponse(url="http://example.com", body=body, encoding='utf-8')): + + attrs = [] + for x in self.xmliter(r, u'þingflokkur'): + attrs.append((x.xpath('@id').extract(), + x.xpath(u'./skammstafanir/stuttskammstöfun/text()').extract(), + x.xpath(u'./tímabil/fyrstaþing/text()').extract())) + + self.assertEqual(attrs, + [([u'26'], [u'-'], [u'80']), + ([u'21'], [u'Ab'], [u'76']), + ([u'27'], [u'A'], [u'27'])]) def test_xmliter_text(self): body = u"""onetwo""" @@ -95,11 +151,15 @@ class XmliterTestCase(unittest.TestCase): self.assertRaises(StopIteration, next, iter) + def test_xmliter_objtype_exception(self): + i = self.xmliter(42, 'product') + self.assertRaises(AssertionError, next, i) + def test_xmliter_encoding(self): body = b'\n\n Some Turkish Characters \xd6\xc7\xde\xdd\xd0\xdc \xfc\xf0\xfd\xfe\xe7\xf6\n\n\n' response = XmlResponse('http://www.example.com', body=body) self.assertEqual( - self.xmliter(response, 'item').next().extract(), + next(self.xmliter(response, 'item')).extract(), u'Some Turkish Characters \xd6\xc7\u015e\u0130\u011e\xdc \xfc\u011f\u0131\u015f\xe7\xf6' ) @@ -168,6 +228,9 @@ class LxmlXmliterTestCase(XmliterTestCase): node = next(my_iter) self.assertEqual(node.xpath('f:name/text()').extract(), ['African Coffee Table']) + def test_xmliter_objtype_exception(self): + i = self.xmliter(42, 'product') + self.assertRaises(TypeError, next, i) class UtilsCsvTestCase(unittest.TestCase): sample_feeds_dir = os.path.join(os.path.abspath(os.path.dirname(__file__)), 'sample_data', 'feeds') @@ -189,11 +252,11 @@ class UtilsCsvTestCase(unittest.TestCase): # explicit type check cuz' we no like stinkin' autocasting! yarrr for result_row in result: - self.assert_(all((isinstance(k, unicode) for k in result_row.keys()))) - self.assert_(all((isinstance(v, unicode) for v in result_row.values()))) + self.assert_(all((isinstance(k, six.text_type) for k in result_row.keys()))) + self.assert_(all((isinstance(v, six.text_type) for v in result_row.values()))) def test_csviter_delimiter(self): - body = get_testdata('feeds', 'feed-sample3.csv').replace(',', '\t') + body = get_testdata('feeds', 'feed-sample3.csv').replace(b',', b'\t') response = TextResponse(url="http://example.com/", body=body) csv = csviter(response, delimiter='\t') @@ -205,8 +268,8 @@ class UtilsCsvTestCase(unittest.TestCase): def test_csviter_quotechar(self): body1 = get_testdata('feeds', 'feed-sample6.csv') - body2 = get_testdata('feeds', 'feed-sample6.csv').replace(",", '|') - + body2 = get_testdata('feeds', 'feed-sample6.csv').replace(b',', b'|') + response1 = TextResponse(url="http://example.com/", body=body1) csv1 = csviter(response1, quotechar="'") @@ -237,7 +300,7 @@ class UtilsCsvTestCase(unittest.TestCase): {u"'id'": u"4", u"'name'": u"'empty'", u"'value'": u""}]) def test_csviter_delimiter_binary_response_assume_utf8_encoding(self): - body = get_testdata('feeds', 'feed-sample3.csv').replace(',', '\t') + body = get_testdata('feeds', 'feed-sample3.csv').replace(b',', b'\t') response = Response(url="http://example.com/", body=body) csv = csviter(response, delimiter='\t') @@ -249,10 +312,10 @@ class UtilsCsvTestCase(unittest.TestCase): def test_csviter_headers(self): sample = get_testdata('feeds', 'feed-sample3.csv').splitlines() - headers, body = sample[0].split(','), '\n'.join(sample[1:]) + headers, body = sample[0].split(b','), b'\n'.join(sample[1:]) response = TextResponse(url="http://example.com/", body=body) - csv = csviter(response, headers=headers) + csv = csviter(response, headers=[h.decode('utf-8') for h in headers]) self.assertEqual([row for row in csv], [{u'id': u'1', u'name': u'alpha', u'value': u'foobar'}, @@ -262,7 +325,7 @@ class UtilsCsvTestCase(unittest.TestCase): def test_csviter_falserow(self): body = get_testdata('feeds', 'feed-sample3.csv') - body = '\n'.join((body, 'a,b', 'a,b,c,d')) + body = b'\n'.join((body, b'a,b', b'a,b,c,d')) response = TextResponse(url="http://example.com/", body=body) csv = csviter(response) diff --git a/tests/test_webclient.py b/tests/test_webclient.py index e0b46286a..9b5beda4c 100644 --- a/tests/test_webclient.py +++ b/tests/test_webclient.py @@ -3,10 +3,11 @@ from twisted.internet import defer Tests borrowed from the twisted.web.client tests. """ import os +import six from six.moves.urllib.parse import urlparse from twisted.trial import unittest -from twisted.web import server, static, error, util +from twisted.web import server, static, util, resource from twisted.internet import reactor, defer from twisted.test.proto_helpers import StringTransport from twisted.python.filepath import FilePath @@ -14,18 +15,21 @@ from twisted.protocols.policies import WrappingFactory from scrapy.core.downloader import webclient as client from scrapy.http import Request, Headers +from scrapy.utils.python import to_bytes, to_unicode -def getPage(url, contextFactory=None, *args, **kwargs): +def getPage(url, contextFactory=None, response_transform=None, *args, **kwargs): """Adapted version of twisted.web.client.getPage""" - def _clientfactory(*args, **kwargs): + def _clientfactory(url, *args, **kwargs): + url = to_unicode(url) timeout = kwargs.pop('timeout', 0) - f = client.ScrapyHTTPClientFactory(Request(*args, **kwargs), timeout=timeout) - f.deferred.addCallback(lambda r: r.body) + f = client.ScrapyHTTPClientFactory( + Request(url, *args, **kwargs), timeout=timeout) + f.deferred.addCallback(response_transform or (lambda r: r.body)) return f from twisted.web.client import _makeGetterFactory - return _makeGetterFactory(url, _clientfactory, + return _makeGetterFactory(to_bytes(url), _clientfactory, contextFactory=contextFactory, *args, **kwargs).deferred @@ -64,6 +68,8 @@ class ParseUrlTestCase(unittest.TestCase): ) for url, test in tests: + test = tuple( + to_bytes(x) if not isinstance(x, int) else x for x in test) self.assertEquals(client._parse(url), test, url) def test_externalUnicodeInterference(self): @@ -72,9 +78,12 @@ class ParseUrlTestCase(unittest.TestCase): elements of its return tuple, even when passed an URL which has previously been passed to L{urlparse} as a C{unicode} string. """ + if not six.PY2: + raise unittest.SkipTest( + "Applies only to Py2, as urls can be ONLY unicode on Py3") badInput = u'http://example.com/path' goodInput = badInput.encode('ascii') - urlparse(badInput) + self._parse(badInput) # cache badInput in urlparse_cached scheme, netloc, host, port, path = self._parse(goodInput) self.assertTrue(isinstance(scheme, str)) self.assertTrue(isinstance(netloc, str)) @@ -99,22 +108,22 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): 'Useful': 'value'})) self._test(factory, - "GET /bar HTTP/1.0\r\n" - "Content-Length: 9\r\n" - "Useful: value\r\n" - "Connection: close\r\n" - "User-Agent: fooble\r\n" - "Host: example.net\r\n" - "Cookie: blah blah\r\n" - "\r\n" - "some data") + b"GET /bar HTTP/1.0\r\n" + b"Content-Length: 9\r\n" + b"Useful: value\r\n" + b"Connection: close\r\n" + b"User-Agent: fooble\r\n" + b"Host: example.net\r\n" + b"Cookie: blah blah\r\n" + b"\r\n" + b"some data") # test minimal sent headers factory = client.ScrapyHTTPClientFactory(Request('http://foo/bar')) self._test(factory, - "GET /bar HTTP/1.0\r\n" - "Host: foo\r\n" - "\r\n") + b"GET /bar HTTP/1.0\r\n" + b"Host: foo\r\n" + b"\r\n") # test a simple POST with body and content-type factory = client.ScrapyHTTPClientFactory(Request( @@ -124,13 +133,13 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): headers={'Content-Type': 'application/x-www-form-urlencoded'})) self._test(factory, - "POST /bar HTTP/1.0\r\n" - "Host: foo\r\n" - "Connection: close\r\n" - "Content-Type: application/x-www-form-urlencoded\r\n" - "Content-Length: 10\r\n" - "\r\n" - "name=value") + b"POST /bar HTTP/1.0\r\n" + b"Host: foo\r\n" + b"Connection: close\r\n" + b"Content-Type: application/x-www-form-urlencoded\r\n" + b"Content-Length: 10\r\n" + b"\r\n" + b"name=value") # test a POST method with no body provided factory = client.ScrapyHTTPClientFactory(Request( @@ -139,10 +148,10 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): )) self._test(factory, - "POST /bar HTTP/1.0\r\n" - "Host: foo\r\n" - "Content-Length: 0\r\n" - "\r\n") + b"POST /bar HTTP/1.0\r\n" + b"Host: foo\r\n" + b"Content-Length: 0\r\n" + b"\r\n") # test with single and multivalued headers factory = client.ScrapyHTTPClientFactory(Request( @@ -153,12 +162,12 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): })) self._test(factory, - "GET /bar HTTP/1.0\r\n" - "Host: foo\r\n" - "X-Meta-Multivalued: value1\r\n" - "X-Meta-Multivalued: value2\r\n" - "X-Meta-Single: single\r\n" - "\r\n") + b"GET /bar HTTP/1.0\r\n" + b"Host: foo\r\n" + b"X-Meta-Multivalued: value1\r\n" + b"X-Meta-Multivalued: value2\r\n" + b"X-Meta-Single: single\r\n" + b"\r\n") # same test with single and multivalued headers but using Headers class factory = client.ScrapyHTTPClientFactory(Request( @@ -169,12 +178,12 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): }))) self._test(factory, - "GET /bar HTTP/1.0\r\n" - "Host: foo\r\n" - "X-Meta-Multivalued: value1\r\n" - "X-Meta-Multivalued: value2\r\n" - "X-Meta-Single: single\r\n" - "\r\n") + b"GET /bar HTTP/1.0\r\n" + b"Host: foo\r\n" + b"X-Meta-Multivalued: value1\r\n" + b"X-Meta-Multivalued: value2\r\n" + b"X-Meta-Single: single\r\n" + b"\r\n") def _test(self, factory, testvalue): transport = StringTransport() @@ -193,10 +202,10 @@ class ScrapyHTTPPageGetterTests(unittest.TestCase): protocol = client.ScrapyHTTPPageGetter() protocol.factory = factory protocol.headers = Headers() - protocol.dataReceived("HTTP/1.0 200 OK\n") - protocol.dataReceived("Hello: World\n") - protocol.dataReceived("Foo: Bar\n") - protocol.dataReceived("\n") + protocol.dataReceived(b"HTTP/1.0 200 OK\n") + protocol.dataReceived(b"Hello: World\n") + protocol.dataReceived(b"Foo: Bar\n") + protocol.dataReceived(b"\n") self.assertEqual(protocol.headers, Headers({'Hello': ['World'], 'Foo': ['Bar']})) @@ -205,6 +214,16 @@ from twisted.web.test.test_webclient import ForeverTakingResource, \ ErrorResource, NoLengthResource, HostHeaderResource, \ PayloadResource, BrokenDownloadResource + +class EncodingResource(resource.Resource): + out_encoding = 'cp1251' + + def render(self, request): + body = to_unicode(request.content.read()) + request.setHeader(b'content-encoding', self.out_encoding) + return body.encode(self.out_encoding) + + class WebClientTestCase(unittest.TestCase): def _listen(self, site): return reactor.listenTCP(0, site, interface="127.0.0.1") @@ -212,15 +231,16 @@ class WebClientTestCase(unittest.TestCase): def setUp(self): name = self.mktemp() os.mkdir(name) - FilePath(name).child("file").setContent("0123456789") + FilePath(name).child("file").setContent(b"0123456789") r = static.File(name) - r.putChild("redirect", util.Redirect("/file")) - r.putChild("wait", ForeverTakingResource()) - r.putChild("error", ErrorResource()) - r.putChild("nolength", NoLengthResource()) - r.putChild("host", HostHeaderResource()) - r.putChild("payload", PayloadResource()) - r.putChild("broken", BrokenDownloadResource()) + r.putChild(b"redirect", util.Redirect(b"/file")) + r.putChild(b"wait", ForeverTakingResource()) + r.putChild(b"error", ErrorResource()) + r.putChild(b"nolength", NoLengthResource()) + r.putChild(b"host", HostHeaderResource()) + r.putChild(b"payload", PayloadResource()) + r.putChild(b"broken", BrokenDownloadResource()) + r.putChild(b"encoding", EncodingResource()) self.site = server.Site(r, timeout=None) self.wrapper = WrappingFactory(self.site) self.port = self._listen(self.wrapper) @@ -234,14 +254,17 @@ class WebClientTestCase(unittest.TestCase): def testPayload(self): s = "0123456789" * 10 - return getPage(self.getURL("payload"), body=s).addCallback(self.assertEquals, s) + return getPage(self.getURL("payload"), body=s).addCallback( + self.assertEquals, to_bytes(s)) def testHostHeader(self): # if we pass Host header explicitly, it should be used, otherwise # it should extract from url return defer.gatherResults([ - getPage(self.getURL("host")).addCallback(self.assertEquals, "127.0.0.1:%d" % self.portno), - getPage(self.getURL("host"), headers={"Host": "www.example.com"}).addCallback(self.assertEquals, "www.example.com")]) + getPage(self.getURL("host")).addCallback( + self.assertEquals, to_bytes("127.0.0.1:%d" % self.portno)), + getPage(self.getURL("host"), headers={"Host": "www.example.com"}).addCallback( + self.assertEquals, to_bytes("www.example.com"))]) def test_getPage(self): @@ -250,7 +273,7 @@ class WebClientTestCase(unittest.TestCase): the body of the response if the default method B{GET} is used. """ d = getPage(self.getURL("file")) - d.addCallback(self.assertEquals, "0123456789") + d.addCallback(self.assertEquals, b"0123456789") return d @@ -263,8 +286,8 @@ class WebClientTestCase(unittest.TestCase): def _getPage(method): return getPage(self.getURL("file"), method=method) return defer.gatherResults([ - _getPage("head").addCallback(self.assertEqual, ""), - _getPage("HEAD").addCallback(self.assertEqual, "")]) + _getPage("head").addCallback(self.assertEqual, b""), + _getPage("HEAD").addCallback(self.assertEqual, b"")]) def test_timeoutNotTriggering(self): @@ -274,7 +297,8 @@ class WebClientTestCase(unittest.TestCase): called back with the contents of the page. """ d = getPage(self.getURL("host"), timeout=100) - d.addCallback(self.assertEquals, "127.0.0.1:%d" % self.portno) + d.addCallback( + self.assertEquals, to_bytes("127.0.0.1:%d" % self.portno)) return d @@ -290,7 +314,7 @@ class WebClientTestCase(unittest.TestCase): def cleanup(passthrough): # Clean up the server which is hanging around not doing # anything. - connected = self.wrapper.protocols.keys() + connected = list(six.iterkeys(self.wrapper.protocols)) # There might be nothing here if the server managed to already see # that the connection was lost. if connected: @@ -303,26 +327,40 @@ class WebClientTestCase(unittest.TestCase): return getPage(self.getURL('notsuchfile')).addCallback(self._cbNoSuchFile) def _cbNoSuchFile(self, pageData): - self.assert_('404 - No Such Resource' in pageData) + self.assert_(b'404 - No Such Resource' in pageData) def testFactoryInfo(self): url = self.getURL('file') - scheme, netloc, host, port, path = client._parse(url) + _, _, host, port, _ = client._parse(url) factory = client.ScrapyHTTPClientFactory(Request(url)) - reactor.connectTCP(host, port, factory) + reactor.connectTCP(to_unicode(host), port, factory) return factory.deferred.addCallback(self._cbFactoryInfo, factory) def _cbFactoryInfo(self, ignoredResult, factory): - self.assertEquals(factory.status, '200') - self.assert_(factory.version.startswith('HTTP/')) - self.assertEquals(factory.message, 'OK') - self.assertEquals(factory.response_headers['content-length'], '10') + self.assertEquals(factory.status, b'200') + self.assert_(factory.version.startswith(b'HTTP/')) + self.assertEquals(factory.message, b'OK') + self.assertEquals(factory.response_headers[b'content-length'], b'10') def testRedirect(self): return getPage(self.getURL("redirect")).addCallback(self._cbRedirect) def _cbRedirect(self, pageData): self.assertEquals(pageData, - '\n\n \n \n' - ' \n \n ' - 'click here\n \n\n') + b'\n\n \n \n' + b' \n \n ' + b'click here\n \n\n') + + def test_Encoding(self): + """ Test that non-standart body encoding matches + Content-Encoding header """ + body = b'\xd0\x81\xd1\x8e\xd0\xaf' + return getPage( + self.getURL('encoding'), body=body, response_transform=lambda r: r)\ + .addCallback(self._check_Encoding, body) + + def _check_Encoding(self, response, original_body): + content_encoding = to_unicode(response.headers[b'Content-Encoding']) + self.assertEquals(content_encoding, EncodingResource.out_encoding) + self.assertEquals( + response.body.decode(content_encoding), to_unicode(original_body)) diff --git a/tox.ini b/tox.ini index eae7e8e47..b8d45d5b9 100644 --- a/tox.ini +++ b/tox.ini @@ -48,6 +48,10 @@ deps = basepython = python3.4 deps = {[testenv:py33]deps} +[testenv:py35] +basepython = python3.5 +deps = {[testenv:py33]deps} + [docs] changedir = docs deps =