diff --git a/conftest.py b/conftest.py
index d5d61ddd3..d54ce155c 100644
--- a/conftest.py
+++ b/conftest.py
@@ -1,4 +1,3 @@
-import six
import pytest
@@ -8,11 +7,10 @@ collect_ignore = [
]
-if six.PY3:
- for line in open('tests/py3-ignores.txt'):
- file_path = line.strip()
- if file_path and file_path[0] != '#':
- collect_ignore.append(file_path)
+for line in open('tests/ignores.txt'):
+ file_path = line.strip()
+ if file_path and file_path[0] != '#':
+ collect_ignore.append(file_path)
@pytest.fixture()
diff --git a/pytest.ini b/pytest.ini
index 7be5d8572..03fcdda5b 100644
--- a/pytest.ini
+++ b/pytest.ini
@@ -214,6 +214,7 @@ flake8-ignore =
tests/test_pipeline_files.py F401 E501 W293 E303 E272 E226
tests/test_pipeline_images.py F401 F841 E501 E303
tests/test_pipeline_media.py E501 E741 E731 E128 E261 E306 E502
+ tests/test_proxy_connect.py E501 E741
tests/test_request_cb_kwargs.py E501
tests/test_responsetypes.py E501 E305
tests/test_robotstxt_interface.py F401 E501 W291 E501
diff --git a/tests/py3-ignores.txt b/tests/ignores.txt
similarity index 74%
rename from tests/py3-ignores.txt
rename to tests/ignores.txt
index 313e74ec9..45cf6fb92 100644
--- a/tests/py3-ignores.txt
+++ b/tests/ignores.txt
@@ -1,6 +1,3 @@
-tests/test_linkextractors_deprecated.py
-tests/test_proxy_connect.py
-
scrapy/linkextractors/sgml.py
scrapy/linkextractors/regex.py
scrapy/linkextractors/htmlparser.py
diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt
index 2e8d319d2..c4bc1f278 100644
--- a/tests/requirements-py3.txt
+++ b/tests/requirements-py3.txt
@@ -1,5 +1,7 @@
# Tests requirements
jmespath
+mitmproxy; python_version >= '3.6'
+mitmproxy==3.0.4; python_version < '3.6'
pytest
pytest-cov
pytest-twisted
diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py
index 59d4a3eec..a78762a64 100644
--- a/tests/test_downloader_handlers.py
+++ b/tests/test_downloader_handlers.py
@@ -1,5 +1,4 @@
import os
-import six
import shutil
import tempfile
from unittest import mock
@@ -630,18 +629,16 @@ class Http11MockServerTestCase(unittest.TestCase):
# download_maxsize < 100, hence the CancelledError
self.assertIsInstance(failure.value, defer.CancelledError)
- if six.PY2:
- request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate')
- request = request.replace(url=self.mockserver.url('/xpayload'))
- yield crawler.crawl(seed=request)
- # download_maxsize = 50 is enough for the gzipped response
- failure = crawler.spider.meta.get('failure')
- self.assertTrue(failure is None)
- reason = crawler.spider.meta['close_reason']
- self.assertTrue(reason, 'finished')
- else:
- # See issue https://twistedmatrix.com/trac/ticket/8175
- raise unittest.SkipTest("xpayload only enabled for PY2")
+ # See issue https://twistedmatrix.com/trac/ticket/8175
+ raise unittest.SkipTest("xpayload fails on PY3")
+ request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate')
+ request = request.replace(url=self.mockserver.url('/xpayload'))
+ yield crawler.crawl(seed=request)
+ # download_maxsize = 50 is enough for the gzipped response
+ failure = crawler.spider.meta.get('failure')
+ self.assertTrue(failure is None)
+ reason = crawler.spider.meta['close_reason']
+ self.assertTrue(reason, 'finished')
class UriResource(resource.Resource):
diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py
deleted file mode 100644
index 1366971be..000000000
--- a/tests/test_linkextractors_deprecated.py
+++ /dev/null
@@ -1,233 +0,0 @@
-# -*- coding: utf-8 -*-
-import unittest
-from scrapy.linkextractors.regex import RegexLinkExtractor
-from scrapy.http import HtmlResponse
-from scrapy.link import Link
-from scrapy.linkextractors.htmlparser import HtmlParserLinkExtractor
-from scrapy.linkextractors.sgml import SgmlLinkExtractor, BaseSgmlLinkExtractor
-from tests import get_testdata
-
-from tests.test_linkextractors import Base
-
-
-class BaseSgmlLinkExtractorTestCase(unittest.TestCase):
- # XXX: should we move some of these tests to base link extractor tests?
-
- def test_basic(self):
- html = """
Page title
- Item 12
- About us
-
- Other category
- >>
-
- """
- response = HtmlResponse("http://example.org/somepage/index.html", body=html)
-
- lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href
- self.assertEqual(lx.extract_links(response),
- [Link(url='http://example.org/somepage/item/12.html', text='Item 12'),
- Link(url='http://example.org/about.html', text='About us'),
- Link(url='http://example.org/othercat.html', text='Other category'),
- Link(url='http://example.org/', text='>>'),
- Link(url='http://example.org/', text='')])
-
- def test_base_url(self):
- html = """Page title
- Item 12
- """
- response = HtmlResponse("http://example.org/somepage/index.html", body=html)
-
- lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href
- self.assertEqual(lx.extract_links(response),
- [Link(url='http://otherdomain.com/base/item/12.html', text='Item 12')])
-
- # base url is an absolute path and relative to host
- html = """Page title
- Item 12
"""
- response = HtmlResponse("https://example.org/somepage/index.html", body=html)
- self.assertEqual(lx.extract_links(response),
- [Link(url='https://example.org/item/12.html', text='Item 12')])
-
- # base url has no scheme
- html = """Page title
- Item 12
"""
- response = HtmlResponse("https://example.org/somepage/index.html", body=html)
- self.assertEqual(lx.extract_links(response),
- [Link(url='https://noschemedomain.com/path/to/item/12.html', text='Item 12')])
-
- def test_link_text_wrong_encoding(self):
- html = """Wrong: \xed
"""
- response = HtmlResponse("http://www.example.com", body=html, encoding='utf-8')
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://www.example.com/item/12.html', text=u'Wrong: \ufffd'),
- ])
-
- def test_extraction_encoding(self):
- body = get_testdata('link_extractor', 'linkextractor_noenc.html')
- response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']})
- response_noenc = HtmlResponse(url='http://example.com/noenc', body=body)
- body = get_testdata('link_extractor', 'linkextractor_latin1.html')
- response_latin1 = HtmlResponse(url='http://example.com/latin1', body=body)
-
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.extract_links(response_utf8), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')),
- ])
-
- self.assertEqual(lx.extract_links(response_noenc), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')),
- ])
-
- # document encoding does not affect URL path component, only query part
- # >>> u'sample_ñ.html'.encode('utf8')
- # b'sample_\xc3\xb1.html'
- # >>> u"sample_á.html".encode('utf8')
- # b'sample_\xc3\xa1.html'
- # >>> u"sample_ö.html".encode('utf8')
- # b'sample_\xc3\xb6.html'
- # >>> u"£32".encode('latin1')
- # b'\xa332'
- # >>> u"µ".encode('latin1')
- # b'\xb5'
- self.assertEqual(lx.extract_links(response_latin1), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%C3%A1.html', text='sample \xe1 text'.decode('latin1')),
- Link(url='http://example.com/sample_%C3%B6.html?price=%A332&%B5=unit', text=''),
- ])
-
- def test_matches(self):
- url1 = 'http://lotsofstuff.com/stuff1/index'
- url2 = 'http://evenmorestuff.com/uglystuff/index'
-
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.matches(url1), True)
- self.assertEqual(lx.matches(url2), True)
-
-
-class HtmlParserLinkExtractorTestCase(unittest.TestCase):
-
- def setUp(self):
- body = get_testdata('link_extractor', 'sgml_linkextractor.html')
- self.response = HtmlResponse(url='http://example.com/index', body=body)
-
- def test_extraction(self):
- # Default arguments
- lx = HtmlParserLinkExtractor()
- self.assertEqual(lx.extract_links(self.response), [
- Link(url='http://example.com/sample2.html', text=u'sample 2'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 text'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 repetition'),
- Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'),
- Link(url='http://www.google.com/something', text=u''),
- Link(url='http://example.com/innertag.html', text=u'inner tag'),
- Link(url='http://example.com/page%204.html', text=u'href with whitespaces'),
- ])
-
- def test_link_wrong_href(self):
- html = """
- Item 1
- Item 2
- Item 3
- """
- response = HtmlResponse("http://example.org/index.html", body=html)
- lx = HtmlParserLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False),
- Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False),
- ])
-
-
-class SgmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
- extractor_cls = SgmlLinkExtractor
- escapes_whitespace = True
-
- def test_deny_extensions(self):
- html = """asd and """
- response = HtmlResponse("http://example.org/", body=html)
- lx = SgmlLinkExtractor(deny_extensions="jpg")
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://example.org/page.html', text=u'asd'),
- ])
-
- def test_attrs_sgml(self):
- html = """
- sample text 2"""
- response = HtmlResponse("http://example.com/index.html", body=html)
- lx = SgmlLinkExtractor(attrs="href")
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://example.com/sample1.html', text=u''),
- ])
-
- def test_link_nofollow(self):
- html = """
- Printer-friendly page
- About us
- Something
- """
- response = HtmlResponse("http://example.org/page.html", body=html)
- lx = SgmlLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/page.html?action=print', text=u'Printer-friendly page', nofollow=True),
- Link(url='http://example.org/about.html', text=u'About us', nofollow=False),
- Link(url='http://google.com/something', text=u'Something', nofollow=True),
- ])
-
-
-class RegexLinkExtractorTestCase(unittest.TestCase):
- # XXX: RegexLinkExtractor is not deprecated yet, but it must be rewritten
- # not to depend on SgmlLinkExractor. Its speed is also much worse
- # than it should be.
-
- def setUp(self):
- body = get_testdata('link_extractor', 'sgml_linkextractor.html')
- self.response = HtmlResponse(url='http://example.com/index', body=body)
-
- def test_extraction(self):
- # Default arguments
- lx = RegexLinkExtractor()
- self.assertEqual(lx.extract_links(self.response),
- [Link(url='http://example.com/sample2.html', text=u'sample 2'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 text'),
- Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'),
- Link(url='http://www.google.com/something', text=u''),
- Link(url='http://example.com/innertag.html', text=u'inner tag'),])
-
- def test_link_wrong_href(self):
- html = """
- Item 1
- Item 2
- Item 3
- """
- response = HtmlResponse("http://example.org/index.html", body=html)
- lx = RegexLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False),
- Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False),
- ])
-
- def test_html_base_href(self):
- html = """
-
-
-
-
-
-
-
-
- """
- response = HtmlResponse("http://a.com/", body=html)
- lx = RegexLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://b.com/test.html', text=u'', nofollow=False),
- ])
-
- @unittest.expectedFailure
- def test_extraction(self):
- # RegexLinkExtractor doesn't parse URLs with leading/trailing
- # whitespaces correctly.
- super(RegexLinkExtractorTestCase, self).test_extraction()
diff --git a/tests/test_proxy_connect.py b/tests/test_proxy_connect.py
index ae1236bcb..277455751 100644
--- a/tests/test_proxy_connect.py
+++ b/tests/test_proxy_connect.py
@@ -1,38 +1,53 @@
import json
import os
-import time
+import re
+from subprocess import Popen, PIPE
+import sys
+import pytest
from six.moves.urllib.parse import urlsplit, urlunsplit
-from threading import Thread
-from libmproxy import controller, proxy
-from netlib import http_auth
from testfixtures import LogCapture
from twisted.internet import defer
from twisted.trial.unittest import TestCase
+
from scrapy.utils.test import get_crawler
from scrapy.http import Request
from tests.spiders import SimpleSpider, SingleRequestSpider
from tests.mockserver import MockServer
-class HTTPSProxy(controller.Master, Thread):
+class MitmProxy:
+ auth_user = 'scrapy'
+ auth_pass = 'scrapy'
- def __init__(self):
- password_manager = http_auth.PassManSingleUser('scrapy', 'scrapy')
- authenticator = http_auth.BasicProxyAuth(password_manager, "mitmproxy")
+ def start(self):
+ from scrapy.utils.test import get_testenv
+ script = """
+import sys
+from mitmproxy.tools.main import mitmdump
+sys.argv[0] = "mitmdump"
+sys.exit(mitmdump())
+ """
cert_path = os.path.join(os.path.abspath(os.path.dirname(__file__)),
- 'keys', 'mitmproxy-ca.pem')
- server = proxy.ProxyServer(proxy.ProxyConfig(
- authenticator = authenticator,
- cacert = cert_path),
- 0)
- self.server = server
- Thread.__init__(self)
- controller.Master.__init__(self, server)
+ 'keys', 'mitmproxy-ca.pem')
+ self.proc = Popen([sys.executable,
+ '-c', script,
+ '--listen-host', '127.0.0.1',
+ '--listen-port', '0',
+ '--proxyauth', '%s:%s' % (self.auth_user, self.auth_pass),
+ '--certs', cert_path,
+ '--ssl-insecure',
+ ],
+ stdout=PIPE, env=get_testenv())
+ line = self.proc.stdout.readline().decode('utf-8')
+ host_port = re.search(r'listening at http://([^:]+:\d+)', line).group(1)
+ address = 'http://%s:%s@%s' % (self.auth_user, self.auth_pass, host_port)
+ return address
- def http_address(self):
- return 'http://scrapy:scrapy@%s:%d' % self.server.socket.getsockname()
+ def stop(self):
+ self.proc.kill()
+ self.proc.communicate()
def _wrong_credentials(proxy_url):
@@ -40,6 +55,7 @@ def _wrong_credentials(proxy_url):
bad_auth_proxy[1] = bad_auth_proxy[1].replace('scrapy:scrapy@', 'wrong:wronger@')
return urlunsplit(bad_auth_proxy)
+
class ProxyConnectTestCase(TestCase):
def setUp(self):
@@ -47,17 +63,14 @@ class ProxyConnectTestCase(TestCase):
self.mockserver.__enter__()
self._oldenv = os.environ.copy()
- self._proxy = HTTPSProxy()
- self._proxy.start()
-
- # Wait for the proxy to start.
- time.sleep(1.0)
- os.environ['https_proxy'] = self._proxy.http_address()
- os.environ['http_proxy'] = self._proxy.http_address()
+ self._proxy = MitmProxy()
+ proxy_url = self._proxy.start()
+ os.environ['https_proxy'] = proxy_url
+ os.environ['http_proxy'] = proxy_url
def tearDown(self):
self.mockserver.__exit__(None, None, None)
- self._proxy.shutdown()
+ self._proxy.stop()
os.environ = self._oldenv
@defer.inlineCallbacks
@@ -67,15 +80,7 @@ class ProxyConnectTestCase(TestCase):
yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
self._assert_got_response_code(200, l)
- @defer.inlineCallbacks
- def test_https_noconnect(self):
- proxy = os.environ['https_proxy']
- os.environ['https_proxy'] = proxy + '?noconnect'
- crawler = get_crawler(SimpleSpider)
- with LogCapture() as l:
- yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
- self._assert_got_response_code(200, l)
-
+ @pytest.mark.xfail(reason='Python 3.6+ fails this earlier', condition=sys.version_info.minor >= 6)
@defer.inlineCallbacks
def test_https_connect_tunnel_error(self):
crawler = get_crawler(SimpleSpider)
@@ -100,9 +105,27 @@ class ProxyConnectTestCase(TestCase):
with LogCapture() as l:
yield crawler.crawl(seed=request)
self._assert_got_response_code(200, l)
- echo = json.loads(crawler.spider.meta['responses'][0].body)
+ echo = json.loads(crawler.spider.meta['responses'][0].text)
self.assertTrue('Proxy-Authorization' not in echo['headers'])
+ # The noconnect mode isn't supported by the current mitmproxy, it returns
+ # "Invalid request scheme: https" as it doesn't seem to support full URLs in GET at all,
+ # and it's not clear what behavior is intended by Scrapy and by mitmproxy here.
+ # https://github.com/mitmproxy/mitmproxy/issues/848 may be related.
+ # The Scrapy noconnect mode was required, at least in the past, to work with Crawlera,
+ # and https://github.com/scrapy-plugins/scrapy-crawlera/pull/44 seems to be related.
+
+ @pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests')
+ @defer.inlineCallbacks
+ def test_https_noconnect(self):
+ proxy = os.environ['https_proxy']
+ os.environ['https_proxy'] = proxy + '?noconnect'
+ crawler = get_crawler(SimpleSpider)
+ with LogCapture() as l:
+ yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
+ self._assert_got_response_code(200, l)
+
+ @pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests')
@defer.inlineCallbacks
def test_https_noconnect_auth_error(self):
os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy']) + '?noconnect'
diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py
index 76ee8f48b..2d27d4b81 100644
--- a/tests/test_utils_python.py
+++ b/tests/test_utils_python.py
@@ -9,7 +9,7 @@ from warnings import catch_warnings
from scrapy.utils.python import (
memoizemethod_noargs, binary_is_text, equal_attributes,
- WeakKeyCache, stringify_dict, get_func_args, to_bytes, to_unicode,
+ WeakKeyCache, get_func_args, to_bytes, to_unicode,
without_none_values, MutableChain)
__doctests__ = ['scrapy.utils.python']
@@ -168,33 +168,6 @@ class UtilsPythonTestCase(unittest.TestCase):
gc.collect()
self.assertFalse(len(wk._weakdict))
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict(self):
- d = {'a': 123, u'b': b'c', u'd': u'e', object(): u'e'}
- d2 = stringify_dict(d, keys_only=False)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()))
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.values()))
-
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict_tuples(self):
- tuples = [('a', 123), (u'b', 'c'), (u'd', u'e'), (object(), u'e')]
- d = dict(tuples)
- d2 = stringify_dict(tuples, keys_only=False)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()), d2.keys())
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.values()))
-
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict_keys_only(self):
- d = {'a': 123, u'b': 'c', u'd': u'e', object(): u'e'}
- d2 = stringify_dict(d)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()))
-
def test_get_func_args(self):
def f1(a, b, c):
pass
diff --git a/tests/test_webclient.py b/tests/test_webclient.py
index a81946490..7b015ff8d 100644
--- a/tests/test_webclient.py
+++ b/tests/test_webclient.py
@@ -78,26 +78,6 @@ class ParseUrlTestCase(unittest.TestCase):
to_bytes(x) if not isinstance(x, int) else x for x in test)
self.assertEqual(client._parse(url), test, url)
- def test_externalUnicodeInterference(self):
- """
- L{client._parse} should return C{str} for the scheme, host, and path
- elements of its return tuple, even when passed an URL which has
- previously been passed to L{urlparse} as a C{unicode} string.
- """
- if not six.PY2:
- raise unittest.SkipTest(
- "Applies only to Py2, as urls can be ONLY unicode on Py3")
- badInput = u'http://example.com/path'
- goodInput = badInput.encode('ascii')
- self._parse(badInput) # cache badInput in urlparse_cached
- scheme, netloc, host, port, path = self._parse(goodInput)
- self.assertTrue(isinstance(scheme, str))
- self.assertTrue(isinstance(netloc, str))
- self.assertTrue(isinstance(host, str))
- self.assertTrue(isinstance(path, str))
- self.assertTrue(isinstance(port, int))
-
-
class ScrapyHTTPPageGetterTests(unittest.TestCase):