Merge pull request #4458 from scrapy/azure-pipelines

Set up CI with Azure Pipelines
This commit is contained in:
Mikhail Korobov 2020-07-16 18:09:31 +05:00 committed by GitHub
commit 0f2f1acf04
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
16 changed files with 203 additions and 43 deletions

24
azure-pipelines.yml Normal file
View File

@ -0,0 +1,24 @@
variables:
TOXENV: py
pool:
vmImage: 'windows-latest'
strategy:
matrix:
Python35:
python.version: '3.5'
TOXENV: windows-pinned
Python36:
python.version: '3.6'
Python37:
python.version: '3.7'
Python38:
python.version: '3.8'
steps:
- task: UsePythonVersion@0
inputs:
versionSpec: '$(python.version)'
displayName: 'Use Python $(python.version)'
- script: |
pip install -U tox twine wheel codecov
tox
displayName: 'Run test suite'

View File

@ -83,25 +83,54 @@ def add_http_if_no_scheme(url):
return url
def _is_posix_path(string):
return bool(
re.match(
r'''
^ # start with...
(
\. # ...a single dot,
(
\. | [^/\.]+ # optionally followed by
)? # either a second dot or some characters
|
~ # $HOME
)? # optional match of ".", ".." or ".blabla"
/ # at least one "/" for a file path,
. # and something after the "/"
''',
string,
flags=re.VERBOSE,
)
)
def _is_windows_path(string):
return bool(
re.match(
r'''
^
(
[a-z]:\\
| \\\\
)
''',
string,
flags=re.IGNORECASE | re.VERBOSE,
)
)
def _is_filesystem_path(string):
return _is_posix_path(string) or _is_windows_path(string)
def guess_scheme(url):
"""Add an URL scheme if missing: file:// for filepath-like input or http:// otherwise."""
parts = urlparse(url)
if parts.scheme:
return url
# Note: this does not match Windows filepath
if re.match(r'''^ # start with...
(
\. # ...a single dot,
(
\. | [^/\.]+ # optionally followed by
)? # either a second dot or some characters
)? # optional match of ".", ".." or ".blabla"
/ # at least one "/" for a file path,
. # and something after the "/"
''', parts.path, flags=re.VERBOSE):
"""Add an URL scheme if missing: file:// for filepath-like input or
http:// otherwise."""
if _is_filesystem_path(url):
return any_to_uri(url)
else:
return add_http_if_no_scheme(url)
return add_http_if_no_scheme(url)
def strip_url(url, strip_credentials=True, strip_default_port=True, origin_only=False, strip_fragment=True):

View File

@ -1,7 +1,9 @@
from urllib.parse import urlparse
from twisted.internet import reactor
from twisted.names.client import createResolver
from twisted.names import cache, hosts as hostsModule, resolve
from twisted.names.client import Resolver
from twisted.python.runtime import platform
from scrapy import Spider, Request
from scrapy.crawler import CrawlerRunner
@ -10,6 +12,16 @@ from scrapy.utils.log import configure_logging
from tests.mockserver import MockServer, MockDNSServer
# https://stackoverflow.com/a/32784190
def createResolver(servers=None, resolvconf=None, hosts=None):
if hosts is None:
hosts = b'/etc/hosts' if platform.getType() == 'posix' else r'c:\windows\hosts'
theResolver = Resolver(resolvconf, servers)
hostResolver = hostsModule.Resolver(hosts)
chain = [hostResolver, cache.CacheResolver(), theResolver]
return resolve.ResolverChain(chain)
class LocalhostSpider(Spider):
name = "localhost_spider"

View File

@ -218,9 +218,8 @@ class MockServer:
self.proc.communicate()
def url(self, path, is_secure=False):
host = self.http_address.replace('0.0.0.0', '127.0.0.1')
if is_secure:
host = self.https_address
host = self.https_address if is_secure else self.http_address
host = host.replace('0.0.0.0', '127.0.0.1')
return host + path
@ -248,9 +247,8 @@ class MockDNSServer:
def __enter__(self):
self.proc = Popen([sys.executable, '-u', '-m', 'tests.mockserver', '-t', 'dns'],
stdout=PIPE, env=get_testenv())
host, port = self.proc.stdout.readline().strip().decode('ascii').split(":")
self.host = host
self.port = int(port)
self.host = '127.0.0.1'
self.port = int(self.proc.stdout.readline().strip().decode('ascii').split(":")[1])
return self
def __exit__(self, exc_type, exc_value, traceback):

View File

@ -5,10 +5,11 @@ mitmproxy; python_version >= '3.6'
mitmproxy<4.0.0; python_version < '3.6'
# https://github.com/pytest-dev/pytest-twisted/issues/93
pytest != 5.4, != 5.4.1
pytest-azurepipelines
pytest-cov
pytest-twisted >= 1.11
pytest-xdist
sybil
sybil >= 1.3.0 # https://github.com/cjw296/sybil/issues/20#issuecomment-605433422
testfixtures
# optional for shell wrapper tests

View File

@ -96,7 +96,7 @@ class ShellTest(ProcessTest, SiteTest, unittest.TestCase):
@defer.inlineCallbacks
def test_local_file(self):
filepath = join(tests_datadir, 'test_site/index.html')
filepath = join(tests_datadir, 'test_site', 'index.html')
_, out, _ = yield self.execute([filepath, '-c', 'item'])
assert b'{}' in out

View File

@ -2,7 +2,7 @@ import inspect
import json
import optparse
import os
from stat import S_IWRITE as ANYONE_WRITE_PERMISSION
import platform
import subprocess
import sys
import tempfile
@ -11,8 +11,10 @@ from itertools import chain
from os.path import exists, join, abspath
from pathlib import Path
from shutil import rmtree, copytree
from stat import S_IWRITE as ANYONE_WRITE_PERMISSION
from tempfile import mkdtemp
from threading import Timer
from unittest import skipIf
from twisted.trial import unittest
@ -489,6 +491,9 @@ class BadSpider(scrapy.Spider):
self.assertIn("start_requests", log)
self.assertIn("badspider.py", log)
# https://twistedmatrix.com/trac/ticket/9766
@skipIf(platform.system() == 'Windows' and sys.version_info >= (3, 8),
"the asyncio reactor is broken on Windows when running Python ≥ 3.8")
def test_asyncio_enabled_true(self):
log = self.get_log(self.debug_log_spider, args=[
'-s', 'TWISTED_REACTOR=twisted.internet.asyncioreactor.AsyncioSelectorReactor'

View File

@ -1,8 +1,10 @@
import logging
import os
import platform
import subprocess
import sys
import warnings
from unittest import skipIf
from pytest import raises, mark
from testfixtures import LogCapture
@ -251,6 +253,9 @@ class CrawlerRunnerHasSpider(unittest.TestCase):
})
@defer.inlineCallbacks
# https://twistedmatrix.com/trac/ticket/9766
@skipIf(platform.system() == 'Windows' and sys.version_info >= (3, 8),
"the asyncio reactor is broken on Windows when running Python ≥ 3.8")
def test_crawler_process_asyncio_enabled_true(self):
with LogCapture(level=logging.DEBUG) as log:
if self.reactor_pytest == 'asyncio':
@ -292,11 +297,17 @@ class CrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
self.assertIn('Spider closed (finished)', log)
self.assertNotIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
# https://twistedmatrix.com/trac/ticket/9766
@skipIf(platform.system() == 'Windows' and sys.version_info >= (3, 8),
"the asyncio reactor is broken on Windows when running Python ≥ 3.8")
def test_asyncio_enabled_no_reactor(self):
log = self.run_script('asyncio_enabled_no_reactor.py')
self.assertIn('Spider closed (finished)', log)
self.assertIn("Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor", log)
# https://twistedmatrix.com/trac/ticket/9766
@skipIf(platform.system() == 'Windows' and sys.version_info >= (3, 8),
"the asyncio reactor is broken on Windows when running Python ≥ 3.8")
def test_asyncio_enabled_reactor(self):
log = self.run_script('asyncio_enabled_reactor.py')
self.assertIn('Spider closed (finished)', log)
@ -320,11 +331,15 @@ class CrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
self.assertIn("Spider closed (finished)", log)
self.assertIn("Using reactor: twisted.internet.selectreactor.SelectReactor", log)
@mark.skipif(platform.system() == 'Windows', reason="PollReactor is not supported on Windows")
def test_reactor_poll(self):
log = self.run_script("twisted_reactor_poll.py")
self.assertIn("Spider closed (finished)", log)
self.assertIn("Using reactor: twisted.internet.pollreactor.PollReactor", log)
# https://twistedmatrix.com/trac/ticket/9766
@skipIf(platform.system() == 'Windows' and sys.version_info >= (3, 8),
"the asyncio reactor is broken on Windows when running Python ≥ 3.8")
def test_reactor_asyncio(self):
log = self.run_script("twisted_reactor_asyncio.py")
self.assertIn("Spider closed (finished)", log)

View File

@ -455,9 +455,15 @@ class FeedExportTest(unittest.TestCase):
def run_and_export(self, spider_cls, settings):
""" Run spider with specified settings; return exported data. """
def path_to_url(path):
return urljoin('file:', pathname2url(str(path)))
def printf_escape(string):
return string.replace('%', '%%')
FEEDS = settings.get('FEEDS') or {}
settings['FEEDS'] = {
urljoin('file:', pathname2url(str(file_path))): feed
printf_escape(path_to_url(file_path)): feed
for file_path, feed in FEEDS.items()
}

View File

@ -1,5 +1,6 @@
import json
import os
import platform
import re
import sys
from subprocess import Popen, PIPE
@ -59,6 +60,8 @@ def _wrong_credentials(proxy_url):
@skipIf(sys.version_info < (3, 5, 4),
"requires mitmproxy < 3.0.0, which these tests do not support")
@skipIf(platform.system() == 'Windows' and sys.version_info < (3, 7),
"mitmproxy does not support Windows when running Python < 3.7")
class ProxyConnectTestCase(TestCase):
def setUp(self):

View File

@ -20,13 +20,20 @@ from scrapy.crawler import CrawlerRunner
module_dir = os.path.dirname(os.path.abspath(__file__))
def _copytree(source, target):
try:
shutil.copytree(source, target)
except shutil.Error:
pass
class SpiderLoaderTest(unittest.TestCase):
def setUp(self):
orig_spiders_dir = os.path.join(module_dir, 'test_spiders')
self.tmpdir = tempfile.mkdtemp()
self.spiders_dir = os.path.join(self.tmpdir, 'test_spiders_xxx')
shutil.copytree(orig_spiders_dir, self.spiders_dir)
_copytree(orig_spiders_dir, self.spiders_dir)
sys.path.append(self.tmpdir)
settings = Settings({'SPIDER_MODULES': ['test_spiders_xxx']})
self.spider_loader = SpiderLoader.from_settings(settings)
@ -124,7 +131,7 @@ class DuplicateSpiderNameLoaderTest(unittest.TestCase):
self.tmpdir = self.mktemp()
os.mkdir(self.tmpdir)
self.spiders_dir = os.path.join(self.tmpdir, 'test_spiders_xxx')
shutil.copytree(orig_spiders_dir, self.spiders_dir)
_copytree(orig_spiders_dir, self.spiders_dir)
sys.path.append(self.tmpdir)
self.settings = Settings({'SPIDER_MODULES': ['test_spiders_xxx']})
@ -134,8 +141,8 @@ class DuplicateSpiderNameLoaderTest(unittest.TestCase):
def test_dupename_warning(self):
# copy 1 spider module so as to have duplicate spider name
shutil.copyfile(os.path.join(self.tmpdir, 'test_spiders_xxx/spider3.py'),
os.path.join(self.tmpdir, 'test_spiders_xxx/spider3dupe.py'))
shutil.copyfile(os.path.join(self.tmpdir, 'test_spiders_xxx', 'spider3.py'),
os.path.join(self.tmpdir, 'test_spiders_xxx', 'spider3dupe.py'))
with warnings.catch_warnings(record=True) as w:
spider_loader = SpiderLoader.from_settings(self.settings)
@ -156,10 +163,10 @@ class DuplicateSpiderNameLoaderTest(unittest.TestCase):
def test_multiple_dupename_warning(self):
# copy 2 spider modules so as to have duplicate spider name
# This should issue 2 warning, 1 for each duplicate spider name
shutil.copyfile(os.path.join(self.tmpdir, 'test_spiders_xxx/spider1.py'),
os.path.join(self.tmpdir, 'test_spiders_xxx/spider1dupe.py'))
shutil.copyfile(os.path.join(self.tmpdir, 'test_spiders_xxx/spider2.py'),
os.path.join(self.tmpdir, 'test_spiders_xxx/spider2dupe.py'))
shutil.copyfile(os.path.join(self.tmpdir, 'test_spiders_xxx', 'spider1.py'),
os.path.join(self.tmpdir, 'test_spiders_xxx', 'spider1dupe.py'))
shutil.copyfile(os.path.join(self.tmpdir, 'test_spiders_xxx', 'spider2.py'),
os.path.join(self.tmpdir, 'test_spiders_xxx', 'spider2dupe.py'))
with warnings.catch_warnings(record=True) as w:
spider_loader = SpiderLoader.from_settings(self.settings)

View File

@ -1,4 +1,6 @@
from unittest import TestCase
import platform
import sys
from unittest import skipIf, TestCase
from pytest import mark
@ -12,6 +14,9 @@ class AsyncioTest(TestCase):
# the result should depend only on the pytest --reactor argument
self.assertEqual(is_asyncio_reactor_installed(), self.reactor_pytest == 'asyncio')
# https://twistedmatrix.com/trac/ticket/9766
@skipIf(platform.system() == 'Windows' and sys.version_info >= (3, 8),
"the asyncio reactor is broken on Windows when running Python ≥ 3.8")
def test_install_asyncio_reactor(self):
# this should do nothing
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")

View File

@ -7,7 +7,7 @@ from scrapy.http import XmlResponse, TextResponse, Response
from tests import get_testdata
FOOBAR_NL = u"foo\nbar"
FOOBAR_NL = "foo{}bar".format(os.linesep)
class XmliterTestCase(unittest.TestCase):

View File

@ -1,7 +1,10 @@
import unittest
from io import StringIO
from time import sleep, time
from unittest import mock
from twisted.trial.unittest import SkipTest
from scrapy.utils import trackref
@ -55,7 +58,18 @@ Foo 1 oldest: 0s ago\n\n''')
def test_get_oldest(self):
o1 = Foo() # NOQA
o1_time = time()
o2 = Bar() # NOQA
o3_time = time()
if o3_time <= o1_time:
sleep(0.01)
o3_time = time()
if o3_time <= o1_time:
raise SkipTest('time.time is not precise enough')
o3 = Foo() # NOQA
self.assertIs(trackref.get_oldest('Foo'), o1)
self.assertIs(trackref.get_oldest('Bar'), o2)

View File

@ -1,8 +1,14 @@
import unittest
from scrapy.spiders import Spider
from scrapy.utils.url import (url_is_from_any_domain, url_is_from_spider,
add_http_if_no_scheme, guess_scheme, strip_url)
from scrapy.utils.url import (
add_http_if_no_scheme,
guess_scheme,
_is_filesystem_path,
strip_url,
url_is_from_any_domain,
url_is_from_spider,
)
__doctests__ = ['scrapy.utils.url']
@ -433,5 +439,28 @@ class StripUrl(unittest.TestCase):
self.assertEqual(strip_url(i, origin_only=True), o)
class IsPathTestCase(unittest.TestCase):
def test_path(self):
for input_value, output_value in (
# https://en.wikipedia.org/wiki/Path_(computing)#Representations_of_paths_by_operating_system_and_shell
# Unix-like OS, Microsoft Windows / cmd.exe
("/home/user/docs/Letter.txt", True),
("./inthisdir", True),
("../../greatgrandparent", True),
("~/.rcinfo", True),
(r"C:\user\docs\Letter.txt", True),
("/user/docs/Letter.txt", True),
(r"C:\Letter.txt", True),
(r"\\Server01\user\docs\Letter.txt", True),
(r"\\?\UNC\Server01\user\docs\Letter.txt", True),
(r"\\?\C:\user\docs\Letter.txt", True),
(r"C:\user\docs\somefile.ext:alternate_stream_name", True),
(r"https://example.com", False),
):
self.assertEqual(_is_filesystem_path(input_value), output_value, input_value)
if __name__ == "__main__":
unittest.main()

18
tox.ini
View File

@ -63,14 +63,12 @@ basepython = pypy3
commands =
py.test {posargs:--durations=10 docs scrapy tests}
[testenv:pinned]
basepython = python3
[pinned]
deps =
-ctests/constraints.txt
cryptography==2.0
cssselect==0.9.1
itemadapter==0.1.0
lxml==3.5.0
parsel==1.5.0
Protego==0.1.15
PyDispatcher==2.0.5
@ -85,6 +83,20 @@ deps =
botocore==1.3.23
Pillow==3.4.2
[testenv:pinned]
basepython = python3
deps =
{[pinned]deps}
lxml==3.5.0
[testenv:windows-pinned]
basepython = python3
deps =
{[pinned]deps}
# First lxml version that includes a Windows wheel for Python 3.5, so we do
# not need to build lxml from sources in a CI Windows job:
lxml==3.8.0
[testenv:extra-deps]
deps =
{[testenv]deps}