revert f-string changes for files under sep/

Issue #4324
This commit is contained in:
Ammar Najjar 2020-08-26 13:49:39 +02:00
parent 7597193dbe
commit 2ca8dfb4b0
No known key found for this signature in database
GPG Key ID: 4E358D38135EE7EA
4 changed files with 29 additions and 28 deletions

View File

@ -30,7 +30,7 @@ Proposed Implementation
if hasattr(value, '__iter__'): # str/unicode not allowed
return [self._field.to_python(v) for v in value]
else:
raise TypeError(f"Expected iterable, got {type(value).__name__}")
raise TypeError("Expected iterable, got %s" % type(value).__name__)
def get_default(self):
# must return a new copy to avoid unexpected behaviors with mutable defaults

View File

@ -11,7 +11,7 @@ SEP-004: Library API
====================
.. note:: the library API has been implemented, but slightly different from
proposed in this SEP. You can run a Scrapy crawler inside a Twisted
reactor, but not outside it.
reactor, but not outside it.
Introduction
============
@ -49,7 +49,7 @@ Here's a simple proof-of-concept code of such script:
cr = Crawler(start_urls, callback=parse_start_page)
cr.run() # blocking call - this populates scraped_items
print(f"{len(scraped_items)} items scraped")
print "%d items scraped" % len(scraped_items)
# ... do something more interesting with scraped_items ...
The behaviour of the Scrapy crawler would be controller by the Scrapy settings,

View File

@ -21,7 +21,7 @@ Current flaws and inconsistencies
2. Link extractors are inflexible and hard to maintain, link
processing/filtering is tightly coupled. (e.g. canonicalize)
3. Isn't possible to crawl an url directly from command line because the Spider
does not know which callback use.
does not know which callback use.
These flaws will be corrected by the changes proposed in this SEP.
@ -55,7 +55,7 @@ Request Extractors
Request Extractors takes response object and determines which requests follow.
This is an enhancement to ``LinkExtractors`` which returns urls (links),
Request Extractors return Request objects.
Request Extractors return Request objects.
Request Processors
------------------
@ -142,7 +142,7 @@ Custom Processor and External Callback
# Callback defined out of spider
def my_external_callback(response):
# process item
# process item
pass
class SampleSpider(CrawlSpider):
@ -233,7 +233,7 @@ Request/Response Matchers
def matches_request(self, request):
"""Returns True if Request's url matches initial url"""
return self.matches_url(request.url)
return self.matches_url(request.url)
def matches_response(self, response):
"""REturns True if Response's url matches initial url"""
@ -305,14 +305,14 @@ Request Extractor
for req in self.requests:
req.meta.setdefault('link_text', '')
req.meta['link_text'] = str_to_unicode(req.meta['link_text'],
encoding)
encoding)
def reset(self):
"""Reset state"""
FixedSGMLParser.reset(self)
self.requests = []
self.base_url = None
def unknown_starttag(self, tag, attrs):
"""Process unknown start tag"""
if 'base' tag:
@ -376,7 +376,7 @@ Request Processor
#!python
#
# Request Processors
# Request Processors
# Processors receive list of requests and return list of requests
#
"""Request Processors"""
@ -390,7 +390,7 @@ Request Processor
# replace in-place
req.url = canonicalize_url(req.url)
yield req
class Unique(object):
"""Filter duplicate Requests"""
@ -455,9 +455,9 @@ Request Processor
"""Initialize allow/deny attributes"""
_re_type = type(re.compile('', 0))
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
self.allow_res = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(allow)]
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
self.deny_res = [x if isinstance(x, _re_type) else re.compile(x)
for x in arg_to_iter(deny)]
def __call__(self, requests):
@ -524,7 +524,7 @@ Rules Manager
#
# Handles rules matcher/callbacks
# Resolve rule for given response
#
#
class RulesManager(object):
"""Rules Manager"""
def __init__(self, rules, spider, default_matcher=UrlRegexMatcher):
@ -542,8 +542,8 @@ Rules Manager
# instance default matcher
matcher = default_matcher(rule.matcher)
else:
raise ValueError('Not valid matcher given '
f'{rule.matcher!r} in {rule!r}')
raise ValueError('Not valid matcher given %r in %r' \
% (rule.matcher, rule))
# prepare callback
if callable(rule.callback):
@ -553,7 +553,8 @@ Rules Manager
callback = getattr(spider, rule.callback)
if not callable(callback):
raise AttributeError(f'Invalid callback {callback!r} can not be resolved')
raise AttributeError('Invalid callback %r can not be resolved' \
% callback)
else:
callback = None

View File

@ -171,7 +171,7 @@ the same spider:
#!python
class MySpider(BaseSpider):
middlewares = [RegexLinkExtractor(), CallbackRules(), CanonicalizeUrl(),
middlewares = [RegexLinkExtractor(), CallbackRules(), CanonicalizeUrl(),
ItemIdSetter(), OffsiteMiddleware()]
allowed_domains = ['example.com', 'sub.example.com']
@ -196,7 +196,7 @@ the same spider:
# extract item from response
return item
The Spider Middleware that implements spider code
The Spider Middleware that implements spider code
=================================================
There's gonna be one middleware that will take care of calling the proper
@ -324,7 +324,7 @@ Another example could be for building URL canonicalizers:
class CanonializeUrl(object):
def process_request(self, request, response, spider):
curl = canonicalize_url(request.url,
curl = canonicalize_url(request.url,
rules=spider.canonicalization_rules)
return request.replace(url=curl)
@ -332,7 +332,7 @@ Another example could be for building URL canonicalizers:
class MySpider(BaseSpider):
middlewares = [CanonicalizeUrl()]
canonicalization_rules = ['sort-query-args',
canonicalization_rules = ['sort-query-args',
'normalize-percent-encoding', ...]
# ...
@ -414,7 +414,7 @@ A spider middleware to avoid visiting pages forbidden by robots.txt:
if netloc in info.pending:
res = None
else:
robotsurl = f"{url.scheme}://{netloc}/robots.txt"
robotsurl = "%s://%s/robots.txt" % (url.scheme, netloc)
meta = {'spider': spider, {'handle_httpstatus_list': [403, 404, 500]}
res = Request(robotsurl, callback=self.parse_robots,
meta=meta, priority=self.REQUEST_PRIORITY)
@ -474,7 +474,7 @@ This is a port of the Offsite middleware to the new spider middleware API:
if host and host not in info.hosts_seen:
spider.log("Filtered offsite request to %r: %s" % (host, request))
info.hosts_seen.add(host)
def should_follow(self, request, spider):
info = self.spiders[spider]
# hostname can be None for wrong urls (like javascript links)
@ -484,7 +484,7 @@ This is a port of the Offsite middleware to the new spider middleware API:
def get_host_regex(self, spider):
"""Override this method to implement a different offsite policy"""
domains = [d.replace('.', r'\.') for d in spider.allowed_domains]
regex = fr'^(.*\.)?({"|".join(domains)})$'
regex = r'^(.*\.)?(%s)$' % '|'.join(domains)
return re.compile(regex)
def spider_opened(self, spider):
@ -570,7 +570,7 @@ A middleware to filter out requests already seen:
self.dupefilter = load_object(clspath)()
dispatcher.connect(self.spider_opened, signal=signals.spider_opened)
dispatcher.connect(self.spider_closed, signal=signals.spider_closed)
def enqueue_request(self, spider, request):
seen = self.dupefilter.request_seen(spider, request)
if not seen or request.dont_filter:
@ -601,8 +601,8 @@ A middleware to Scrape data using Parsley as described in UsingParsley
for name in parslet.keys():
self.fields[name] = Field()
super(ParsleyItem, self).__init__(*a, **kw)
self.item_class = ParsleyItem
self.parsley = PyParsley(parslet, output='python')
self.item_class = ParsleyItem
self.parsley = PyParsley(parslet, output='python')
def process_response(self, response, request, spider):
return self.item_class(self.parsly.parse(string=response.body))
@ -627,7 +627,7 @@ Resolved:
not the original one (think of redirections), but it does carry the ``meta``
of the original one. The original one may not be available anymore (in
memory) if we're using a persistent scheduler., but in that case it would be
the deserialized request from the persistent scheduler queue.
the deserialized request from the persistent scheduler queue.
- No - this would make implementation more complex and we're not sure it's
really needed