From bc9db65358bcae3fc228d4b8099d319cba479ff2 Mon Sep 17 00:00:00 2001 From: Leonid Amirov Date: Mon, 2 Nov 2015 16:08:19 +0300 Subject: [PATCH] issue GH #1550 - scrapy shell argument fixes: "example.com" requests "http://example.com"; "example" requests "file://example"; "./example.com" requests "file://example.com" --- scrapy/commands/shell.py | 13 ++++++++++++- 1 file changed, 12 insertions(+), 1 deletion(-) diff --git a/scrapy/commands/shell.py b/scrapy/commands/shell.py index f10da4370..cb441bc9d 100644 --- a/scrapy/commands/shell.py +++ b/scrapy/commands/shell.py @@ -5,11 +5,13 @@ See documentation in docs/topics/shell.rst """ from threading import Thread +import urlparse from w3lib.url import any_to_uri from scrapy.commands import ScrapyCommand from scrapy.shell import Shell from scrapy.http import Request +from scrapy.utils.url import add_http_if_no_scheme from scrapy.utils.spider import spidercls_for_request, DefaultSpider @@ -43,7 +45,16 @@ class Command(ScrapyCommand): def run(self, args, opts): url = args[0] if args else None if url: - url = any_to_uri(url) + parts = urlparse.urlsplit(url) + if not parts.scheme: + if "." not in parts.path.split("/", 1)[0]: + url = any_to_uri(url) + + for pattern in ["/", "./", "../"]: + if url.startswith(pattern): + url = any_to_uri(url) + break + url = add_http_if_no_scheme(url) spider_loader = self.crawler_process.spider_loader