Allow the parse command to write data to a file (#4377)

This commit is contained in:
Aditya Kumar 2020-07-11 12:18:24 +05:30 committed by GitHub
parent 56a6d22352
commit 3f7e8635f4
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
4 changed files with 34 additions and 19 deletions

View File

@ -468,7 +468,7 @@ Supported options:
* ``--callback`` or ``-c``: spider method to use as callback for parsing the
response
* ``--meta`` or ``-m``: additional request meta that will be passed to the callback
* ``--meta`` or ``-m``: additional request meta that will be passed to the callback
request. This must be a valid json string. Example: --meta='{"foo" : "bar"}'
* ``--cbkwargs``: additional keyword arguments that will be passed to the callback.
@ -491,6 +491,8 @@ Supported options:
* ``--verbose`` or ``-v``: display information for each depth level
* ``--output`` or ``-o``: dump scraped items to a file
.. skip: start
Usage example::

View File

@ -108,7 +108,7 @@ class ScrapyCommand:
class BaseRunSpiderCommand(ScrapyCommand):
"""
Common class used to share functionality between the crawl and runspider commands
Common class used to share functionality between the crawl, parse and runspider commands
"""
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)

View File

@ -4,18 +4,16 @@ import logging
from itemadapter import is_item, ItemAdapter
from w3lib.url import is_url
from scrapy.commands import ScrapyCommand
from scrapy.commands import BaseRunSpiderCommand
from scrapy.http import Request
from scrapy.utils import display
from scrapy.utils.conf import arglist_to_dict
from scrapy.utils.spider import iterate_spider_output, spidercls_for_request
from scrapy.exceptions import UsageError
logger = logging.getLogger(__name__)
class Command(ScrapyCommand):
class Command(BaseRunSpiderCommand):
requires_project = True
spider = None
@ -31,11 +29,9 @@ class Command(ScrapyCommand):
return "Parse URL (using its spider) and print the results"
def add_options(self, parser):
ScrapyCommand.add_options(self, parser)
BaseRunSpiderCommand.add_options(self, parser)
parser.add_option("--spider", dest="spider", default=None,
help="use this spider without looking for one")
parser.add_option("-a", dest="spargs", action="append", default=[], metavar="NAME=VALUE",
help="set spider argument (may be repeated)")
parser.add_option("--pipelines", action="store_true",
help="process items through pipelines")
parser.add_option("--nolinks", dest="nolinks", action="store_true",
@ -200,12 +196,15 @@ class Command(ScrapyCommand):
self.add_items(depth, items)
self.add_requests(depth, requests)
scraped_data = items if opts.output else []
if depth < opts.depth:
for req in requests:
req.meta['_depth'] = depth + 1
req.meta['_callback'] = req.callback
req.callback = callback
return requests
scraped_data += requests
return scraped_data
# update request meta if any extra meta was passed through the --meta/-m opts.
if opts.meta:
@ -221,18 +220,11 @@ class Command(ScrapyCommand):
return request
def process_options(self, args, opts):
ScrapyCommand.process_options(self, args, opts)
BaseRunSpiderCommand.process_options(self, args, opts)
self.process_spider_arguments(opts)
self.process_request_meta(opts)
self.process_request_cb_kwargs(opts)
def process_spider_arguments(self, opts):
try:
opts.spargs = arglist_to_dict(opts.spargs)
except ValueError:
raise UsageError("Invalid -a value, use -a NAME=VALUE", print_help=False)
def process_request_meta(self, opts):
if opts.meta:
try:

View File

@ -1,5 +1,5 @@
import os
from os.path import join, abspath
from os.path import join, abspath, isfile, exists
from twisted.internet import defer
from scrapy.utils.testsite import SiteTest
from scrapy.utils.testproc import ProcessTest
@ -218,3 +218,24 @@ ITEM_PIPELINES = {'%s.pipelines.MyPipeline': 1}
)
self.assertRegex(_textmode(out), r"""# Scraped Items -+\n\[\]""")
self.assertIn("""Cannot find a rule that matches""", _textmode(stderr))
@defer.inlineCallbacks
def test_output_flag(self):
"""Checks if a file was created successfully having
correct format containing correct data in it.
"""
file_name = 'data.json'
file_path = join(self.proj_path, file_name)
yield self.execute([
'--spider', self.spider_name,
'-c', 'parse',
'-o', file_name,
self.url('/html')
])
self.assertTrue(exists(file_path))
self.assertTrue(isfile(file_path))
content = '[\n{},\n{"foo": "bar"}\n]'
with open(file_path, 'r') as f:
self.assertEqual(f.read(), content)