Merge branch 'master' into flake8-remove-e128

This commit is contained in:
Eugenio Lacuesta 2020-05-08 18:33:29 -03:00
commit be39f274b8
No known key found for this signature in database
GPG Key ID: DA3EF2D0913E9810
32 changed files with 248 additions and 111 deletions

View File

@ -11,6 +11,8 @@ matrix:
python: 3.8
- env: TOXENV=flake8
python: 3.8
- env: TOXENV=pylint
python: 3.8
- env: TOXENV=docs
python: 3.7 # Keep in sync with .readthedocs.yml

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
#
# Scrapy documentation build configuration file, created by
# sphinx-quickstart on Mon Nov 24 12:02:52 2008.
#

View File

@ -14,50 +14,57 @@ Author: dufferzafar
import re
# Used for remembering the file (and its contents)
# so we don't have to open the same file again.
_filename = None
_contents = None
# A regex that matches standard linkcheck output lines
line_re = re.compile(u'(.*)\:\d+\:\s\[(.*)\]\s(?:(.*)\sto\s(.*)|(.*))')
def main():
# Read lines from the linkcheck output file
try:
with open("build/linkcheck/output.txt") as out:
output_lines = out.readlines()
except IOError:
print("linkcheck output not found; please run linkcheck first.")
exit(1)
# Used for remembering the file (and its contents)
# so we don't have to open the same file again.
_filename = None
_contents = None
# For every line, fix the respective file
for line in output_lines:
match = re.match(line_re, line)
# A regex that matches standard linkcheck output lines
line_re = re.compile(u'(.*)\:\d+\:\s\[(.*)\]\s(?:(.*)\sto\s(.*)|(.*))')
if match:
newfilename = match.group(1)
errortype = match.group(2)
# Read lines from the linkcheck output file
try:
with open("build/linkcheck/output.txt") as out:
output_lines = out.readlines()
except IOError:
print("linkcheck output not found; please run linkcheck first.")
exit(1)
# Broken links can't be fixed and
# I am not sure what do with the local ones.
if errortype.lower() in ["broken", "local"]:
print("Not Fixed: " + line)
# For every line, fix the respective file
for line in output_lines:
match = re.match(line_re, line)
if match:
newfilename = match.group(1)
errortype = match.group(2)
# Broken links can't be fixed and
# I am not sure what do with the local ones.
if errortype.lower() in ["broken", "local"]:
print("Not Fixed: " + line)
else:
# If this is a new file
if newfilename != _filename:
# Update the previous file
if _filename:
with open(_filename, "w") as _file:
_file.write(_contents)
_filename = newfilename
# Read the new file to memory
with open(_filename) as _file:
_contents = _file.read()
_contents = _contents.replace(match.group(3), match.group(4))
else:
# If this is a new file
if newfilename != _filename:
# We don't understand what the current line means!
print("Not Understood: " + line)
# Update the previous file
if _filename:
with open(_filename, "w") as _file:
_file.write(_contents)
_filename = newfilename
# Read the new file to memory
with open(_filename) as _file:
_contents = _file.read()
_contents = _contents.replace(match.group(3), match.group(4))
else:
# We don't understand what the current line means!
print("Not Understood: " + line)
if __name__ == '__main__':
main()

113
pylintrc Normal file
View File

@ -0,0 +1,113 @@
[MASTER]
persistent=no
jobs=1 # >1 hides results
[MESSAGES CONTROL]
disable=abstract-method,
anomalous-backslash-in-string,
arguments-differ,
attribute-defined-outside-init,
bad-classmethod-argument,
bad-continuation,
bad-indentation,
bad-mcs-classmethod-argument,
bad-super-call,
bad-whitespace,
bare-except,
blacklisted-name,
broad-except,
c-extension-no-member,
catching-non-exception,
cell-var-from-loop,
comparison-with-callable,
consider-iterating-dictionary,
consider-using-in,
consider-using-set-comprehension,
consider-using-sys-exit,
cyclic-import,
dangerous-default-value,
deprecated-method,
deprecated-module,
duplicate-code, # https://github.com/PyCQA/pylint/issues/214
eval-used,
expression-not-assigned,
fixme,
function-redefined,
global-statement,
import-error,
import-outside-toplevel,
import-self,
inconsistent-return-statements,
inherit-non-class,
invalid-name,
invalid-overridden-method,
isinstance-second-argument-not-valid-type,
keyword-arg-before-vararg,
line-too-long,
logging-format-interpolation,
logging-not-lazy,
lost-exception,
method-hidden,
misplaced-comparison-constant,
missing-docstring,
missing-final-newline,
multiple-imports,
multiple-statements,
no-else-continue,
no-else-raise,
no-else-return,
no-init,
no-member,
no-method-argument,
no-name-in-module,
no-self-argument,
no-self-use,
no-value-for-parameter,
not-an-iterable,
not-callable,
pointless-statement,
pointless-string-statement,
protected-access,
redefined-argument-from-local,
redefined-builtin,
redefined-outer-name,
reimported,
signature-differs,
singleton-comparison,
super-init-not-called,
superfluous-parens,
too-few-public-methods,
too-many-ancestors,
too-many-arguments,
too-many-branches,
too-many-format-args,
too-many-function-args,
too-many-instance-attributes,
too-many-lines,
too-many-locals,
too-many-public-methods,
too-many-return-statements,
trailing-newlines,
trailing-whitespace,
unbalanced-tuple-unpacking,
undefined-variable,
undefined-loop-variable,
unexpected-special-method-signature,
ungrouped-imports,
unidiomatic-typecheck,
unnecessary-comprehension,
unnecessary-lambda,
unnecessary-pass,
unreachable,
unsubscriptable-object,
unused-argument,
unused-import,
unused-variable,
unused-wildcard-import,
used-before-assignment,
useless-object-inheritance, # Required for Python 2 support
useless-return,
useless-super-delegation,
wildcard-import,
wrong-import-order,
wrong-import-position

View File

@ -96,15 +96,15 @@ flake8-ignore =
scrapy/loader/processors.py E501
# scrapy/pipelines
scrapy/pipelines/__init__.py E501
scrapy/pipelines/files.py E116 E501
scrapy/pipelines/files.py E501
scrapy/pipelines/images.py E501
scrapy/pipelines/media.py E501
# scrapy/selector
scrapy/selector/__init__.py F403
scrapy/selector/unified.py E501 E111
scrapy/selector/unified.py E501
# scrapy/settings
scrapy/settings/__init__.py E501
scrapy/settings/default_settings.py E501 E114 E116
scrapy/settings/default_settings.py E501
scrapy/settings/deprecated.py E501
# scrapy/spidermiddlewares
scrapy/spidermiddlewares/httperror.py E501
@ -205,7 +205,7 @@ flake8-ignore =
tests/test_item.py E501 F841
tests/test_link.py E501
tests/test_linkextractors.py E501
tests/test_loader.py E501 E741 E117
tests/test_loader.py E501 E741
tests/test_logformatter.py E501
tests/test_mail.py E501
tests/test_middleware.py E501
@ -221,10 +221,10 @@ flake8-ignore =
tests/test_selector.py E501
tests/test_spider.py E501
tests/test_spidermiddleware.py E501
tests/test_spidermiddleware_httperror.py E501 E121
tests/test_spidermiddleware_offsite.py E501 E111
tests/test_spidermiddleware_httperror.py E501
tests/test_spidermiddleware_offsite.py E501
tests/test_spidermiddleware_output_chain.py E501
tests/test_spidermiddleware_referer.py E501 F841 E501 E121
tests/test_spidermiddleware_referer.py E501 F841 E501
tests/test_squeues.py E501 E741
tests/test_utils_asyncio.py E501
tests/test_utils_conf.py E501

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import re
import logging

View File

@ -1,6 +1,8 @@
"""
Link extractor based on lxml.html
"""
import operator
from functools import partial
from urllib.parse import urljoin
import lxml.etree as etree
@ -8,10 +10,10 @@ from w3lib.html import strip_html5_whitespace
from w3lib.url import canonicalize_url, safe_url_string
from scrapy.link import Link
from scrapy.linkextractors import FilteringLinkExtractor
from scrapy.utils.misc import arg_to_iter, rel_has_nofollow
from scrapy.utils.python import unique as unique_list
from scrapy.utils.response import get_base_url
from scrapy.linkextractors import FilteringLinkExtractor
# from lxml/src/lxml/html/__init__.py
@ -27,19 +29,24 @@ def _nons(tag):
return tag
def _identity(x):
return x
def _canonicalize_link_url(link):
return canonicalize_url(link.url, keep_fragments=True)
class LxmlParserLinkExtractor:
def __init__(self, tag="a", attr="href", process=None, unique=False,
strip=True, canonicalized=False):
self.scan_tag = tag if callable(tag) else lambda t: t == tag
self.scan_attr = attr if callable(attr) else lambda a: a == attr
self.process_attr = process if callable(process) else lambda v: v
def __init__(
self, tag="a", attr="href", process=None, unique=False, strip=True, canonicalized=False
):
self.scan_tag = tag if callable(tag) else partial(operator.eq, tag)
self.scan_attr = attr if callable(attr) else partial(operator.eq, attr)
self.process_attr = process if callable(process) else _identity
self.unique = unique
self.strip = strip
if canonicalized:
self.link_key = lambda link: link.url
else:
self.link_key = lambda link: canonicalize_url(link.url,
keep_fragments=True)
self.link_key = operator.attrgetter("url") if canonicalized else _canonicalize_link_url
def _iter_links(self, document):
for el in document.iter(etree.Element):
@ -93,25 +100,44 @@ class LxmlParserLinkExtractor:
class LxmlLinkExtractor(FilteringLinkExtractor):
def __init__(self, allow=(), deny=(), allow_domains=(), deny_domains=(), restrict_xpaths=(),
tags=('a', 'area'), attrs=('href',), canonicalize=False,
unique=True, process_value=None, deny_extensions=None, restrict_css=(),
strip=True, restrict_text=None):
def __init__(
self,
allow=(),
deny=(),
allow_domains=(),
deny_domains=(),
restrict_xpaths=(),
tags=('a', 'area'),
attrs=('href',),
canonicalize=False,
unique=True,
process_value=None,
deny_extensions=None,
restrict_css=(),
strip=True,
restrict_text=None,
):
tags, attrs = set(arg_to_iter(tags)), set(arg_to_iter(attrs))
lx = LxmlParserLinkExtractor(
tag=lambda x: x in tags,
attr=lambda x: x in attrs,
tag=partial(operator.contains, tags),
attr=partial(operator.contains, attrs),
unique=unique,
process=process_value,
strip=strip,
canonicalized=canonicalize
)
super(LxmlLinkExtractor, self).__init__(lx, allow=allow, deny=deny,
allow_domains=allow_domains, deny_domains=deny_domains,
restrict_xpaths=restrict_xpaths, restrict_css=restrict_css,
canonicalize=canonicalize, deny_extensions=deny_extensions,
restrict_text=restrict_text)
super(LxmlLinkExtractor, self).__init__(
link_extractor=lx,
allow=allow,
deny=deny,
allow_domains=allow_domains,
deny_domains=deny_domains,
restrict_xpaths=restrict_xpaths,
restrict_css=restrict_css,
canonicalize=canonicalize,
deny_extensions=deny_extensions,
restrict_text=restrict_text,
)
def extract_links(self, response):
"""Returns a list of :class:`~scrapy.link.Link` objects from the
@ -124,9 +150,11 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
"""
base_url = get_base_url(response)
if self.restrict_xpaths:
docs = [subdoc
for x in self.restrict_xpaths
for subdoc in response.xpath(x)]
docs = [
subdoc
for x in self.restrict_xpaths
for subdoc in response.xpath(x)
]
else:
docs = [response.selector]
all_links = []

View File

@ -83,8 +83,7 @@ class S3FilesStore:
AWS_USE_SSL = None
AWS_VERIFY = None
POLICY = 'private' # Overriden from settings.FILES_STORE_S3_ACL in
# FilesPipeline.from_settings.
POLICY = 'private' # Overriden from settings.FILES_STORE_S3_ACL in FilesPipeline.from_settings
HEADERS = {
'Cache-Control': 'max-age=172800',
}

View File

@ -65,9 +65,9 @@ class Selector(_ParselSelector, object_ref):
selectorlist_cls = SelectorList
def __init__(self, response=None, text=None, type=None, root=None, **kwargs):
if not(response is None or text is None):
raise ValueError('%s.__init__() received both response and text'
% self.__class__.__name__)
if response is not None and text is not None:
raise ValueError('%s.__init__() received both response and text'
% self.__class__.__name__)
st = _st(response, type or self._default_type)

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import traceback
import warnings
from collections import defaultdict

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
# Define here the models for your scraped items
#
# See documentation in:

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
# Define here the models for your spider middleware
#
# See documentation in:

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
# Define your item pipelines here
#
# Don't forget to add your pipeline to the ITEM_PIPELINES setting

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
# Scrapy settings for $project_name project
#
# For simplicity, this file contains only settings considered important or

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import scrapy

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import scrapy
from scrapy.linkextractors import LinkExtractor
from scrapy.spiders import CrawlSpider, Rule

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
from scrapy.spiders import CSVFeedSpider

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
from scrapy.spiders import XMLFeedSpider

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
import logging
import sys
import warnings

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
import OpenSSL
import OpenSSL._util as pyOpenSSLutil

View File

@ -1,5 +1,3 @@
# -*- coding: utf-8 -*-
import unittest
from scrapy.downloadermiddlewares.redirect import RedirectMiddleware, MetaRefreshMiddleware

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
from unittest import mock
from twisted.internet import reactor, error

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import unittest
from w3lib.encoding import resolve_encoding

View File

@ -1,3 +1,4 @@
import pickle
import re
import unittest
from warnings import catch_warnings
@ -462,6 +463,10 @@ class Base:
Link(url='ftp://www.external.com/', text=u'An Item', fragment='', nofollow=False),
])
def test_pickle_extractor(self):
lx = self.extractor_cls()
self.assertIsInstance(pickle.loads(pickle.dumps(lx)), self.extractor_cls)
class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
extractor_cls = LxmlLinkExtractor

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import os
import shutil

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import unittest
from scrapy.responsetypes import responsetypes

View File

@ -21,10 +21,10 @@ class _HttpErrorSpider(MockServerSpider):
def __init__(self, *args, **kwargs):
super(_HttpErrorSpider, self).__init__(*args, **kwargs)
self.start_urls = [
self.mockserver.url("/status?n=200"),
self.mockserver.url("/status?n=404"),
self.mockserver.url("/status?n=402"),
self.mockserver.url("/status?n=500"),
self.mockserver.url("/status?n=200"),
self.mockserver.url("/status?n=404"),
self.mockserver.url("/status?n=402"),
self.mockserver.url("/status?n=500"),
]
self.failed = set()
self.skipped = set()

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import inspect
import unittest
from unittest import mock

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import os
from twisted.trial import unittest

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import sys
import logging
import unittest

View File

@ -1,4 +1,3 @@
# -*- coding: utf-8 -*-
import unittest
from scrapy.spiders import Spider

13
tox.ini
View File

@ -37,6 +37,19 @@ deps =
pytest-flake8
commands =
py.test --flake8 {posargs:docs scrapy tests}
[testenv:pylint]
basepython = python3
deps =
{[testenv]deps}
# Optional dependencies
boto
reppy
robotexclusionrulesparser
# Test dependencies
pylint
commands =
pylint conftest.py docs extras scrapy setup.py tests
[testenv:pypy3]
basepython = pypy3