diff --git a/conftest.py b/conftest.py
index d5d61ddd3..74fb101e9 100644
--- a/conftest.py
+++ b/conftest.py
@@ -1,4 +1,3 @@
-import six
import pytest
@@ -8,11 +7,10 @@ collect_ignore = [
]
-if six.PY3:
- for line in open('tests/py3-ignores.txt'):
- file_path = line.strip()
- if file_path and file_path[0] != '#':
- collect_ignore.append(file_path)
+for line in open('tests/py3-ignores.txt'):
+ file_path = line.strip()
+ if file_path and file_path[0] != '#':
+ collect_ignore.append(file_path)
@pytest.fixture()
diff --git a/tests/py3-ignores.txt b/tests/py3-ignores.txt
index 313e74ec9..45cf6fb92 100644
--- a/tests/py3-ignores.txt
+++ b/tests/py3-ignores.txt
@@ -1,6 +1,3 @@
-tests/test_linkextractors_deprecated.py
-tests/test_proxy_connect.py
-
scrapy/linkextractors/sgml.py
scrapy/linkextractors/regex.py
scrapy/linkextractors/htmlparser.py
diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py
index 59d4a3eec..b06fcf6c3 100644
--- a/tests/test_downloader_handlers.py
+++ b/tests/test_downloader_handlers.py
@@ -630,18 +630,16 @@ class Http11MockServerTestCase(unittest.TestCase):
# download_maxsize < 100, hence the CancelledError
self.assertIsInstance(failure.value, defer.CancelledError)
- if six.PY2:
- request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate')
- request = request.replace(url=self.mockserver.url('/xpayload'))
- yield crawler.crawl(seed=request)
- # download_maxsize = 50 is enough for the gzipped response
- failure = crawler.spider.meta.get('failure')
- self.assertTrue(failure is None)
- reason = crawler.spider.meta['close_reason']
- self.assertTrue(reason, 'finished')
- else:
- # See issue https://twistedmatrix.com/trac/ticket/8175
- raise unittest.SkipTest("xpayload only enabled for PY2")
+ # See issue https://twistedmatrix.com/trac/ticket/8175
+ raise unittest.SkipTest("xpayload fails on PY3")
+ request.headers.setdefault(b'Accept-Encoding', b'gzip,deflate')
+ request = request.replace(url=self.mockserver.url('/xpayload'))
+ yield crawler.crawl(seed=request)
+ # download_maxsize = 50 is enough for the gzipped response
+ failure = crawler.spider.meta.get('failure')
+ self.assertTrue(failure is None)
+ reason = crawler.spider.meta['close_reason']
+ self.assertTrue(reason, 'finished')
class UriResource(resource.Resource):
diff --git a/tests/test_linkextractors_deprecated.py b/tests/test_linkextractors_deprecated.py
deleted file mode 100644
index 1366971be..000000000
--- a/tests/test_linkextractors_deprecated.py
+++ /dev/null
@@ -1,233 +0,0 @@
-# -*- coding: utf-8 -*-
-import unittest
-from scrapy.linkextractors.regex import RegexLinkExtractor
-from scrapy.http import HtmlResponse
-from scrapy.link import Link
-from scrapy.linkextractors.htmlparser import HtmlParserLinkExtractor
-from scrapy.linkextractors.sgml import SgmlLinkExtractor, BaseSgmlLinkExtractor
-from tests import get_testdata
-
-from tests.test_linkextractors import Base
-
-
-class BaseSgmlLinkExtractorTestCase(unittest.TestCase):
- # XXX: should we move some of these tests to base link extractor tests?
-
- def test_basic(self):
- html = """
Page title
- Item 12
- About us
-
- Other category
- >>
-
- """
- response = HtmlResponse("http://example.org/somepage/index.html", body=html)
-
- lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href
- self.assertEqual(lx.extract_links(response),
- [Link(url='http://example.org/somepage/item/12.html', text='Item 12'),
- Link(url='http://example.org/about.html', text='About us'),
- Link(url='http://example.org/othercat.html', text='Other category'),
- Link(url='http://example.org/', text='>>'),
- Link(url='http://example.org/', text='')])
-
- def test_base_url(self):
- html = """Page title
- Item 12
- """
- response = HtmlResponse("http://example.org/somepage/index.html", body=html)
-
- lx = BaseSgmlLinkExtractor() # default: tag=a, attr=href
- self.assertEqual(lx.extract_links(response),
- [Link(url='http://otherdomain.com/base/item/12.html', text='Item 12')])
-
- # base url is an absolute path and relative to host
- html = """Page title
- Item 12
"""
- response = HtmlResponse("https://example.org/somepage/index.html", body=html)
- self.assertEqual(lx.extract_links(response),
- [Link(url='https://example.org/item/12.html', text='Item 12')])
-
- # base url has no scheme
- html = """Page title
- Item 12
"""
- response = HtmlResponse("https://example.org/somepage/index.html", body=html)
- self.assertEqual(lx.extract_links(response),
- [Link(url='https://noschemedomain.com/path/to/item/12.html', text='Item 12')])
-
- def test_link_text_wrong_encoding(self):
- html = """Wrong: \xed
"""
- response = HtmlResponse("http://www.example.com", body=html, encoding='utf-8')
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://www.example.com/item/12.html', text=u'Wrong: \ufffd'),
- ])
-
- def test_extraction_encoding(self):
- body = get_testdata('link_extractor', 'linkextractor_noenc.html')
- response_utf8 = HtmlResponse(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']})
- response_noenc = HtmlResponse(url='http://example.com/noenc', body=body)
- body = get_testdata('link_extractor', 'linkextractor_latin1.html')
- response_latin1 = HtmlResponse(url='http://example.com/latin1', body=body)
-
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.extract_links(response_utf8), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')),
- ])
-
- self.assertEqual(lx.extract_links(response_noenc), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%E2%82%AC.html', text='sample \xe2\x82\xac text'.decode('utf-8')),
- ])
-
- # document encoding does not affect URL path component, only query part
- # >>> u'sample_ñ.html'.encode('utf8')
- # b'sample_\xc3\xb1.html'
- # >>> u"sample_á.html".encode('utf8')
- # b'sample_\xc3\xa1.html'
- # >>> u"sample_ö.html".encode('utf8')
- # b'sample_\xc3\xb6.html'
- # >>> u"£32".encode('latin1')
- # b'\xa332'
- # >>> u"µ".encode('latin1')
- # b'\xb5'
- self.assertEqual(lx.extract_links(response_latin1), [
- Link(url='http://example.com/sample_%C3%B1.html', text=''),
- Link(url='http://example.com/sample_%C3%A1.html', text='sample \xe1 text'.decode('latin1')),
- Link(url='http://example.com/sample_%C3%B6.html?price=%A332&%B5=unit', text=''),
- ])
-
- def test_matches(self):
- url1 = 'http://lotsofstuff.com/stuff1/index'
- url2 = 'http://evenmorestuff.com/uglystuff/index'
-
- lx = BaseSgmlLinkExtractor()
- self.assertEqual(lx.matches(url1), True)
- self.assertEqual(lx.matches(url2), True)
-
-
-class HtmlParserLinkExtractorTestCase(unittest.TestCase):
-
- def setUp(self):
- body = get_testdata('link_extractor', 'sgml_linkextractor.html')
- self.response = HtmlResponse(url='http://example.com/index', body=body)
-
- def test_extraction(self):
- # Default arguments
- lx = HtmlParserLinkExtractor()
- self.assertEqual(lx.extract_links(self.response), [
- Link(url='http://example.com/sample2.html', text=u'sample 2'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 text'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 repetition'),
- Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'),
- Link(url='http://www.google.com/something', text=u''),
- Link(url='http://example.com/innertag.html', text=u'inner tag'),
- Link(url='http://example.com/page%204.html', text=u'href with whitespaces'),
- ])
-
- def test_link_wrong_href(self):
- html = """
- Item 1
- Item 2
- Item 3
- """
- response = HtmlResponse("http://example.org/index.html", body=html)
- lx = HtmlParserLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False),
- Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False),
- ])
-
-
-class SgmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
- extractor_cls = SgmlLinkExtractor
- escapes_whitespace = True
-
- def test_deny_extensions(self):
- html = """asd and """
- response = HtmlResponse("http://example.org/", body=html)
- lx = SgmlLinkExtractor(deny_extensions="jpg")
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://example.org/page.html', text=u'asd'),
- ])
-
- def test_attrs_sgml(self):
- html = """
- sample text 2"""
- response = HtmlResponse("http://example.com/index.html", body=html)
- lx = SgmlLinkExtractor(attrs="href")
- self.assertEqual(lx.extract_links(response), [
- Link(url='http://example.com/sample1.html', text=u''),
- ])
-
- def test_link_nofollow(self):
- html = """
- Printer-friendly page
- About us
- Something
- """
- response = HtmlResponse("http://example.org/page.html", body=html)
- lx = SgmlLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/page.html?action=print', text=u'Printer-friendly page', nofollow=True),
- Link(url='http://example.org/about.html', text=u'About us', nofollow=False),
- Link(url='http://google.com/something', text=u'Something', nofollow=True),
- ])
-
-
-class RegexLinkExtractorTestCase(unittest.TestCase):
- # XXX: RegexLinkExtractor is not deprecated yet, but it must be rewritten
- # not to depend on SgmlLinkExractor. Its speed is also much worse
- # than it should be.
-
- def setUp(self):
- body = get_testdata('link_extractor', 'sgml_linkextractor.html')
- self.response = HtmlResponse(url='http://example.com/index', body=body)
-
- def test_extraction(self):
- # Default arguments
- lx = RegexLinkExtractor()
- self.assertEqual(lx.extract_links(self.response),
- [Link(url='http://example.com/sample2.html', text=u'sample 2'),
- Link(url='http://example.com/sample3.html', text=u'sample 3 text'),
- Link(url='http://example.com/sample3.html#foo', text=u'sample 3 repetition with fragment'),
- Link(url='http://www.google.com/something', text=u''),
- Link(url='http://example.com/innertag.html', text=u'inner tag'),])
-
- def test_link_wrong_href(self):
- html = """
- Item 1
- Item 2
- Item 3
- """
- response = HtmlResponse("http://example.org/index.html", body=html)
- lx = RegexLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://example.org/item1.html', text=u'Item 1', nofollow=False),
- Link(url='http://example.org/item3.html', text=u'Item 3', nofollow=False),
- ])
-
- def test_html_base_href(self):
- html = """
-
-
-
-
-
-
-
-
- """
- response = HtmlResponse("http://a.com/", body=html)
- lx = RegexLinkExtractor()
- self.assertEqual([link for link in lx.extract_links(response)], [
- Link(url='http://b.com/test.html', text=u'', nofollow=False),
- ])
-
- @unittest.expectedFailure
- def test_extraction(self):
- # RegexLinkExtractor doesn't parse URLs with leading/trailing
- # whitespaces correctly.
- super(RegexLinkExtractorTestCase, self).test_extraction()
diff --git a/tests/test_proxy_connect.py b/tests/test_proxy_connect.py
deleted file mode 100644
index ae1236bcb..000000000
--- a/tests/test_proxy_connect.py
+++ /dev/null
@@ -1,120 +0,0 @@
-import json
-import os
-import time
-
-from six.moves.urllib.parse import urlsplit, urlunsplit
-from threading import Thread
-from libmproxy import controller, proxy
-from netlib import http_auth
-from testfixtures import LogCapture
-
-from twisted.internet import defer
-from twisted.trial.unittest import TestCase
-from scrapy.utils.test import get_crawler
-from scrapy.http import Request
-from tests.spiders import SimpleSpider, SingleRequestSpider
-from tests.mockserver import MockServer
-
-
-class HTTPSProxy(controller.Master, Thread):
-
- def __init__(self):
- password_manager = http_auth.PassManSingleUser('scrapy', 'scrapy')
- authenticator = http_auth.BasicProxyAuth(password_manager, "mitmproxy")
- cert_path = os.path.join(os.path.abspath(os.path.dirname(__file__)),
- 'keys', 'mitmproxy-ca.pem')
- server = proxy.ProxyServer(proxy.ProxyConfig(
- authenticator = authenticator,
- cacert = cert_path),
- 0)
- self.server = server
- Thread.__init__(self)
- controller.Master.__init__(self, server)
-
- def http_address(self):
- return 'http://scrapy:scrapy@%s:%d' % self.server.socket.getsockname()
-
-
-def _wrong_credentials(proxy_url):
- bad_auth_proxy = list(urlsplit(proxy_url))
- bad_auth_proxy[1] = bad_auth_proxy[1].replace('scrapy:scrapy@', 'wrong:wronger@')
- return urlunsplit(bad_auth_proxy)
-
-class ProxyConnectTestCase(TestCase):
-
- def setUp(self):
- self.mockserver = MockServer()
- self.mockserver.__enter__()
- self._oldenv = os.environ.copy()
-
- self._proxy = HTTPSProxy()
- self._proxy.start()
-
- # Wait for the proxy to start.
- time.sleep(1.0)
- os.environ['https_proxy'] = self._proxy.http_address()
- os.environ['http_proxy'] = self._proxy.http_address()
-
- def tearDown(self):
- self.mockserver.__exit__(None, None, None)
- self._proxy.shutdown()
- os.environ = self._oldenv
-
- @defer.inlineCallbacks
- def test_https_connect_tunnel(self):
- crawler = get_crawler(SimpleSpider)
- with LogCapture() as l:
- yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
- self._assert_got_response_code(200, l)
-
- @defer.inlineCallbacks
- def test_https_noconnect(self):
- proxy = os.environ['https_proxy']
- os.environ['https_proxy'] = proxy + '?noconnect'
- crawler = get_crawler(SimpleSpider)
- with LogCapture() as l:
- yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
- self._assert_got_response_code(200, l)
-
- @defer.inlineCallbacks
- def test_https_connect_tunnel_error(self):
- crawler = get_crawler(SimpleSpider)
- with LogCapture() as l:
- yield crawler.crawl("https://localhost:99999/status?n=200")
- self._assert_got_tunnel_error(l)
-
- @defer.inlineCallbacks
- def test_https_tunnel_auth_error(self):
- os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy'])
- crawler = get_crawler(SimpleSpider)
- with LogCapture() as l:
- yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
- # The proxy returns a 407 error code but it does not reach the client;
- # he just sees a TunnelError.
- self._assert_got_tunnel_error(l)
-
- @defer.inlineCallbacks
- def test_https_tunnel_without_leak_proxy_authorization_header(self):
- request = Request(self.mockserver.url("/echo", is_secure=True))
- crawler = get_crawler(SingleRequestSpider)
- with LogCapture() as l:
- yield crawler.crawl(seed=request)
- self._assert_got_response_code(200, l)
- echo = json.loads(crawler.spider.meta['responses'][0].body)
- self.assertTrue('Proxy-Authorization' not in echo['headers'])
-
- @defer.inlineCallbacks
- def test_https_noconnect_auth_error(self):
- os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy']) + '?noconnect'
- crawler = get_crawler(SimpleSpider)
- with LogCapture() as l:
- yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
- self._assert_got_response_code(407, l)
-
- def _assert_got_response_code(self, code, log):
- print(log)
- self.assertEqual(str(log).count('Crawled (%d)' % code), 1)
-
- def _assert_got_tunnel_error(self, log):
- print(log)
- self.assertIn('TunnelError', str(log))
diff --git a/tests/test_utils_python.py b/tests/test_utils_python.py
index a94398796..6857356f6 100644
--- a/tests/test_utils_python.py
+++ b/tests/test_utils_python.py
@@ -163,33 +163,6 @@ class UtilsPythonTestCase(unittest.TestCase):
gc.collect()
self.assertFalse(len(wk._weakdict))
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict(self):
- d = {'a': 123, u'b': b'c', u'd': u'e', object(): u'e'}
- d2 = stringify_dict(d, keys_only=False)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()))
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.values()))
-
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict_tuples(self):
- tuples = [('a', 123), (u'b', 'c'), (u'd', u'e'), (object(), u'e')]
- d = dict(tuples)
- d2 = stringify_dict(tuples, keys_only=False)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()), d2.keys())
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.values()))
-
- @unittest.skipUnless(six.PY2, "deprecated function")
- def test_stringify_dict_keys_only(self):
- d = {'a': 123, u'b': 'c', u'd': u'e', object(): u'e'}
- d2 = stringify_dict(d)
- self.assertEqual(d, d2)
- self.assertIsNot(d, d2) # shouldn't modify in place
- self.assertFalse(any(isinstance(x, six.text_type) for x in d2.keys()))
-
def test_get_func_args(self):
def f1(a, b, c):
pass
diff --git a/tests/test_webclient.py b/tests/test_webclient.py
index a81946490..7b015ff8d 100644
--- a/tests/test_webclient.py
+++ b/tests/test_webclient.py
@@ -78,26 +78,6 @@ class ParseUrlTestCase(unittest.TestCase):
to_bytes(x) if not isinstance(x, int) else x for x in test)
self.assertEqual(client._parse(url), test, url)
- def test_externalUnicodeInterference(self):
- """
- L{client._parse} should return C{str} for the scheme, host, and path
- elements of its return tuple, even when passed an URL which has
- previously been passed to L{urlparse} as a C{unicode} string.
- """
- if not six.PY2:
- raise unittest.SkipTest(
- "Applies only to Py2, as urls can be ONLY unicode on Py3")
- badInput = u'http://example.com/path'
- goodInput = badInput.encode('ascii')
- self._parse(badInput) # cache badInput in urlparse_cached
- scheme, netloc, host, port, path = self._parse(goodInput)
- self.assertTrue(isinstance(scheme, str))
- self.assertTrue(isinstance(netloc, str))
- self.assertTrue(isinstance(host, str))
- self.assertTrue(isinstance(path, str))
- self.assertTrue(isinstance(port, int))
-
-
class ScrapyHTTPPageGetterTests(unittest.TestCase):