""" Scrapy Shell See documentation in docs/topics/shell.rst """ from threading import Thread from scrapy.command import ScrapyCommand from scrapy.shell import Shell from scrapy.http import Request from scrapy import log from scrapy.utils.spider import spidercls_for_request, DefaultSpider class Command(ScrapyCommand): requires_project = False default_settings = {'KEEP_ALIVE': True, 'LOGSTATS_INTERVAL': 0} def syntax(self): return "[url|file]" def short_desc(self): return "Interactive scraping console" def long_desc(self): return "Interactive console for scraping the given url" def add_options(self, parser): ScrapyCommand.add_options(self, parser) parser.add_option("-c", dest="code", help="evaluate the code in the shell, print the result and exit") parser.add_option("--spider", dest="spider", help="use this spider") def update_vars(self, vars): """You can use this function to update the Scrapy objects that will be available in the shell """ pass def run(self, args, opts): url = args[0] if args else None spiders = self.crawler_process.spiders spidercls = DefaultSpider if opts.spider: spidercls = spiders.load(opts.spider) elif url: spidercls = spidercls_for_request(spiders, Request(url), spidercls, log_multiple=True) crawler = self.crawler_process._create_logged_crawler(spidercls) crawler.engine = crawler._create_engine() crawler.engine.start() self.crawler_process._start_logging() self._start_crawler_thread() shell = Shell(crawler, update_vars=self.update_vars, code=opts.code) shell.start(url=url) def _start_crawler_thread(self): t = Thread(target=self.crawler_process._start_reactor, kwargs={'stop_after_crawl': False}) t.daemon = True t.start()