diff --git a/conftest.py b/conftest.py
index 5e6a42977..7da4c4976 100644
--- a/conftest.py
+++ b/conftest.py
@@ -6,8 +6,8 @@ collect_ignore = [
"scrapy/utils/testsite.py",
]
-# FIXME: fix or delete these tests
-for line in open('tests/py3-ignores.txt'):
+
+for line in open('tests/ignores.txt'):
file_path = line.strip()
if file_path and file_path[0] != '#':
collect_ignore.append(file_path)
diff --git a/tests/py3-ignores.txt b/tests/ignores.txt
similarity index 74%
rename from tests/py3-ignores.txt
rename to tests/ignores.txt
index 313e74ec9..45cf6fb92 100644
--- a/tests/py3-ignores.txt
+++ b/tests/ignores.txt
@@ -1,6 +1,3 @@
-tests/test_linkextractors_deprecated.py
-tests/test_proxy_connect.py
-
scrapy/linkextractors/sgml.py
scrapy/linkextractors/regex.py
scrapy/linkextractors/htmlparser.py
diff --git a/tests/requirements-py3.txt b/tests/requirements-py3.txt
index dd5b23cc3..c2b16bec6 100644
--- a/tests/requirements-py3.txt
+++ b/tests/requirements-py3.txt
@@ -1,5 +1,7 @@
# Tests requirements
jmespath
+mitmproxy; python_version >= '3.6'
+mitmproxy==3.0.4; python_version < '3.6'
pytest
pytest-cov
pytest-twisted
diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py
index 6090998d4..5e6a2017e 100644
--- a/tests/test_downloader_handlers.py
+++ b/tests/test_downloader_handlers.py
@@ -1,5 +1,4 @@
import os
-import six
import shutil
import tempfile
from unittest import mock
@@ -630,18 +629,16 @@ class Http11MockServerTestCase(unittest.TestCase):
# download_maxsize < 100, hence the CancelledError
self.assertIsInstance(failure.value, defer.CancelledError)
- if six.PY2:
- request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate')
- request = request.replace(url=self.mockserver.url('/xpayload'))
- yield crawler.crawl(seed=request)
- # download_maxsize = 50 is enough for the gzipped response
- failure = crawler.spider.meta.get('failure')
- self.assertTrue(failure == None)
- reason = crawler.spider.meta['close_reason']
- self.assertTrue(reason, 'finished')
- else:
- # See issue https://twistedmatrix.com/trac/ticket/8175
- raise unittest.SkipTest("xpayload only enabled for PY2")
+ # See issue https://twistedmatrix.com/trac/ticket/8175
+ raise unittest.SkipTest("xpayload fails on PY3")
+ request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate')
+ request = request.replace(url=self.mockserver.url('/xpayload'))
+ yield crawler.crawl(seed=request)
+ # download_maxsize = 50 is enough for the gzipped response
+ failure = crawler.spider.meta.get('failure')
+ self.assertTrue(failure == None)
+ reason = crawler.spider.meta['close_reason']
+ self.assertTrue(reason, 'finished')
class UriResource(resource.Resource):
diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py
deleted file mode 100644
index 1366971be..000000000
--- a/tests/test_linkextractors_deprecated.py
+++ /dev/null
@@ -1,233 +0,0 @@
-# -*- coding: utf-8 -*-
-import unittest
-from scrapy.linkextractors.regex import RegexLinkExtractor
-from scrapy.http import HtmlResponse
-from scrapy.link import Link
-from scrapy.linkextractors.htmlparser import HtmlParserLinkExtractor
-from scrapy.linkextractors.sgml import SgmlLinkExtractor, BaseSgmlLinkExtractor
-from tests import get_testdata
-
-from tests.test_linkextractors import Base
-
-
-class BaseSgmlLinkExtractorTestCase(unittest.TestCase):
- # XXX: should we move some of these tests to base link extractor tests?
-
- def test_basic(self):
- html = """
Page title
- Item 12
- About us
-
- Other category
- >>
-
- """
- response = HtmlResponse("http://example.org/somepage/index.html", body=html)
-
- lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href
- self.assertEqual(lx.extract_links(response),
- [Link(url='http://example.org/somepage/item/12.html', text='Item 12'),
- Link(url='http://example.org/about.html', text='About us'),
- Link(url='http://example.org/othercat.html', text='Other category'),
- Link(url='http://example.org/', text='>>'),
- Link(url='http://example.org/', text='')])
-
- def test_base_url(self):
- html = """Page title
- Item 12
- """
- response = HtmlResponse("http://example.org/somepage/index.html", body=html)
-
- lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href
- self.assertEqual(lx.extract_links(response),
- [Link(url='http://otherdomain.com/base/item/12.html', text='Item 12')])
-
- # base url is an absolute path and relative to host
- html = """Page title
- Item 12
"""
- response = HtmlResponse("https://example.org/somepage/index.html", body=html)
- self.assertEqual(lx.extract_links(response),
- [Link(url='https://example.org/item/12.html', text='Item 12')])
-
- # base url has no scheme
- html = """Page title
- Item 12
"""
- response = HtmlResponse("https://example.org/somepage/index.html", body=html)
- self.assertEqual(lx.extract_links(response),
- [Link(url='https://noschemedomain.com/path/to/item/12.html', text='Item 12')])
-
- def test_link_text_wrong_encoding(self):
- html = """Wrong: \xed
"""
- response = HtmlResponse("http://www.example.com", body=html, encoding='utf-8')
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://www.example.com/item/12.html', text=u'Wrong: \ufffd'),
- ])
-
- def test_extraction_encoding(self):
- body = get_testdata('link_extractor', 'linkextractor_noenc.html')
- response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']})
- response_noenc = HtmlResponse(url='http://example.com/noenc', body=body)
- body = get_testdata('link_extractor', 'linkextractor_latin1.html')
- response_latin1 = HtmlResponse(url='http://example.com/latin1', body=body)
-
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.extract_links(response_utf8), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')),
- ])
-
- self.assertEqual(lx.extract_links(response_noenc), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')),
- ])
-
- # document encoding does not affect URL path component, only query part
- # >>> u'sample_ñ.html'.encode('utf8')
- # b'sample_\xc3\xb1.html'
- # >>> u"sample_á.html".encode('utf8')
- # b'sample_\xc3\xa1.html'
- # >>> u"sample_ö.html".encode('utf8')
- # b'sample_\xc3\xb6.html'
- # >>> u"£32".encode('latin1')
- # b'\xa332'
- # >>> u"µ".encode('latin1')
- # b'\xb5'
- self.assertEqual(lx.extract_links(response_latin1), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%C3%A1.html', text='sample \xe1 text'.decode('latin1')),
- Link(url='http://example.com/sample_%C3%B6.html?price=%A332&%B5=unit', text=''),
- ])
-
- def test_matches(self):
- url1 = 'http://lotsofstuff.com/stuff1/index'
- url2 = 'http://evenmorestuff.com/uglystuff/index'
-
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.matches(url1), True)
- self.assertEqual(lx.matches(url2), True)
-
-
-class HtmlParserLinkExtractorTestCase(unittest.TestCase):
-
- def setUp(self):
- body = get_testdata('link_extractor', 'sgml_linkextractor.html')
- self.response = HtmlResponse(url='http://example.com/index', body=body)
-
- def test_extraction(self):
- # Default arguments
- lx = HtmlParserLinkExtractor()
- self.assertEqual(lx.extract_links(self.response), [
- Link(url='http://example.com/sample2.html', text=u'sample 2'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 text'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 repetition'),
- Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'),
- Link(url='http://www.google.com/something', text=u''),
- Link(url='http://example.com/innertag.html', text=u'inner tag'),
- Link(url='http://example.com/page%204.html', text=u'href with whitespaces'),
- ])
-
- def test_link_wrong_href(self):
- html = """
- Item 1
- Item 2
- Item 3
- """
- response = HtmlResponse("http://example.org/index.html", body=html)
- lx = HtmlParserLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False),
- Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False),
- ])
-
-
-class SgmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
- extractor_cls = SgmlLinkExtractor
- escapes_whitespace = True
-
- def test_deny_extensions(self):
- html = """asd and """
- response = HtmlResponse("http://example.org/", body=html)
- lx = SgmlLinkExtractor(deny_extensions="jpg")
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://example.org/page.html', text=u'asd'),
- ])
-
- def test_attrs_sgml(self):
- html = """
- sample text 2"""
- response = HtmlResponse("http://example.com/index.html", body=html)
- lx = SgmlLinkExtractor(attrs="href")
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://example.com/sample1.html', text=u''),
- ])
-
- def test_link_nofollow(self):
- html = """
- Printer-friendly page
- About us
- Something
- """
- response = HtmlResponse("http://example.org/page.html", body=html)
- lx = SgmlLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/page.html?action=print', text=u'Printer-friendly page', nofollow=True),
- Link(url='http://example.org/about.html', text=u'About us', nofollow=False),
- Link(url='http://google.com/something', text=u'Something', nofollow=True),
- ])
-
-
-class RegexLinkExtractorTestCase(unittest.TestCase):
- # XXX: RegexLinkExtractor is not deprecated yet, but it must be rewritten
- # not to depend on SgmlLinkExractor. Its speed is also much worse
- # than it should be.
-
- def setUp(self):
- body = get_testdata('link_extractor', 'sgml_linkextractor.html')
- self.response = HtmlResponse(url='http://example.com/index', body=body)
-
- def test_extraction(self):
- # Default arguments
- lx = RegexLinkExtractor()
- self.assertEqual(lx.extract_links(self.response),
- [Link(url='http://example.com/sample2.html', text=u'sample 2'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 text'),
- Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'),
- Link(url='http://www.google.com/something', text=u''),
- Link(url='http://example.com/innertag.html', text=u'inner tag'),])
-
- def test_link_wrong_href(self):
- html = """
- Item 1
- Item 2
- Item 3
- """
- response = HtmlResponse("http://example.org/index.html", body=html)
- lx = RegexLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False),
- Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False),
- ])
-
- def test_html_base_href(self):
- html = """
-
-
-
-
-
-
-
-
- """
- response = HtmlResponse("http://a.com/", body=html)
- lx = RegexLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://b.com/test.html', text=u'', nofollow=False),
- ])
-
- @unittest.expectedFailure
- def test_extraction(self):
- # RegexLinkExtractor doesn't parse URLs with leading/trailing
- # whitespaces correctly.
- super(RegexLinkExtractorTestCase, self).test_extraction()
diff --git a/tests/test_proxy_connect.py b/tests/test_proxy_connect.py
index 2b62a378d..bf56136b1 100644
--- a/tests/test_proxy_connect.py
+++ b/tests/test_proxy_connect.py
@@ -1,11 +1,11 @@
import json
import os
-import time
+import re
+import sys
from urllib.parse import urlsplit, urlunsplit
-from threading import Thread
+from subprocess import Popen, PIPE
-from libmproxy import controller, proxy
-from netlib import http_auth
+import pytest
from testfixtures import LogCapture
from twisted.internet import defer
from twisted.trial.unittest import TestCase
@@ -16,23 +16,37 @@ from tests.spiders import SimpleSpider, SingleRequestSpider
from tests.mockserver import MockServer
-class HTTPSProxy(controller.Master, Thread):
+class MitmProxy:
+ auth_user = 'scrapy'
+ auth_pass = 'scrapy'
- def __init__(self):
- password_manager = http_auth.PassManSingleUser('scrapy', 'scrapy')
- authenticator = http_auth.BasicProxyAuth(password_manager, "mitmproxy")
+ def start(self):
+ from scrapy.utils.test import get_testenv
+ script = """
+import sys
+from mitmproxy.tools.main import mitmdump
+sys.argv[0] = "mitmdump"
+sys.exit(mitmdump())
+ """
cert_path = os.path.join(os.path.abspath(os.path.dirname(__file__)),
'keys', 'mitmproxy-ca.pem')
- server = proxy.ProxyServer(proxy.ProxyConfig(
- authenticator = authenticator,
- cacert = cert_path),
- 0)
- self.server = server
- Thread.__init__(self)
- controller.Master.__init__(self, server)
+ self.proc = Popen([sys.executable,
+ '-c', script,
+ '--listen-host', '127.0.0.1',
+ '--listen-port', '0',
+ '--proxyauth', '%s:%s' % (self.auth_user, self.auth_pass),
+ '--certs', cert_path,
+ '--ssl-insecure',
+ ],
+ stdout=PIPE, env=get_testenv())
+ line = self.proc.stdout.readline().decode('utf-8')
+ host_port = re.search(r'listening at http://([^:]+:\d+)', line).group(1)
+ address = 'http://%s:%s@%s' % (self.auth_user, self.auth_pass, host_port)
+ return address
- def http_address(self):
- return 'http://scrapy:scrapy@%s:%d' % self.server.socket.getsockname()
+ def stop(self):
+ self.proc.kill()
+ self.proc.communicate()
def _wrong_credentials(proxy_url):
@@ -40,6 +54,7 @@ def _wrong_credentials(proxy_url):
bad_auth_proxy[1] = bad_auth_proxy[1].replace('scrapy:scrapy@', 'wrong:wronger@')
return urlunsplit(bad_auth_proxy)
+
class ProxyConnectTestCase(TestCase):
def setUp(self):
@@ -47,17 +62,14 @@ class ProxyConnectTestCase(TestCase):
self.mockserver.__enter__()
self._oldenv = os.environ.copy()
- self._proxy = HTTPSProxy()
- self._proxy.start()
-
- # Wait for the proxy to start.
- time.sleep(1.0)
- os.environ['https_proxy'] = self._proxy.http_address()
- os.environ['http_proxy'] = self._proxy.http_address()
+ self._proxy = MitmProxy()
+ proxy_url = self._proxy.start()
+ os.environ['https_proxy'] = proxy_url
+ os.environ['http_proxy'] = proxy_url
def tearDown(self):
self.mockserver.__exit__(None, None, None)
- self._proxy.shutdown()
+ self._proxy.stop()
os.environ = self._oldenv
@defer.inlineCallbacks
@@ -67,6 +79,7 @@ class ProxyConnectTestCase(TestCase):
yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
self._assert_got_response_code(200, l)
+ @pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests')
@defer.inlineCallbacks
def test_https_noconnect(self):
proxy = os.environ['https_proxy']
@@ -76,6 +89,7 @@ class ProxyConnectTestCase(TestCase):
yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
self._assert_got_response_code(200, l)
+ @pytest.mark.xfail(reason='Python 3.6+ fails this earlier', condition=sys.version_info.minor >= 6)
@defer.inlineCallbacks
def test_https_connect_tunnel_error(self):
crawler = get_crawler(SimpleSpider)
@@ -100,9 +114,10 @@ class ProxyConnectTestCase(TestCase):
with LogCapture() as l:
yield crawler.crawl(seed=request)
self._assert_got_response_code(200, l)
- echo = json.loads(crawler.spider.meta['responses'][0].body)
+ echo = json.loads(crawler.spider.meta['responses'][0].text)
self.assertTrue('Proxy-Authorization' not in echo['headers'])
+ @pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests')
@defer.inlineCallbacks
def test_https_noconnect_auth_error(self):
os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy']) + '?noconnect'
diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py
index bea431969..faf0d4b73 100644
--- a/tests/test_utils_python.py
+++ b/tests/test_utils_python.py
@@ -2,15 +2,15 @@ import gc
import functools
import operator
import unittest
-from itertools import count
import platform
-import six
+from itertools import count
from scrapy.utils.python import (
memoizemethod_noargs, binary_is_text, equal_attributes,
WeakKeyCache, stringify_dict, get_func_args, to_bytes, to_unicode,
without_none_values, MutableChain)
+
__doctests__ = ['scrapy.utils.python']
@@ -163,33 +163,6 @@ class UtilsPythonTestCase(unittest.TestCase):
gc.collect()
self.assertFalse(len(wk._weakdict))
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict(self):
- d = {'a': 123, u'b': b'c', u'd': u'e', object(): u'e'}
- d2 = stringify_dict(d, keys_only=False)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, str) for x in d2.keys()))
- self.assertFalse(any(isinstance(x, str) for x in d2.values()))
-
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict_tuples(self):
- tuples = [('a', 123), (u'b', 'c'), (u'd', u'e'), (object(), u'e')]
- d = dict(tuples)
- d2 = stringify_dict(tuples, keys_only=False)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, str) for x in d2.keys()), d2.keys())
- self.assertFalse(any(isinstance(x, str) for x in d2.values()))
-
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict_keys_only(self):
- d = {'a': 123, u'b': 'c', u'd': u'e', object(): u'e'}
- d2 = stringify_dict(d)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, str) for x in d2.keys()))
-
def test_get_func_args(self):
def f1(a, b, c):
pass