mirror of https://github.com/scrapy/scrapy.git
Extract domain from genspider URL (#4439)
This commit is contained in:
parent
ada539a63a
commit
3f635eb683
|
|
@ -4,6 +4,7 @@ import string
|
|||
|
||||
from importlib import import_module
|
||||
from os.path import join, dirname, abspath, exists, splitext
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import scrapy
|
||||
from scrapy.commands import ScrapyCommand
|
||||
|
|
@ -22,6 +23,14 @@ def sanitize_module_name(module_name):
|
|||
return module_name
|
||||
|
||||
|
||||
def extract_domain(url):
|
||||
"""Extract domain name from URL string"""
|
||||
o = urlparse(url)
|
||||
if o.scheme == '' and o.netloc == '':
|
||||
o = urlparse("//" + url.lstrip("/"))
|
||||
return o.netloc
|
||||
|
||||
|
||||
class Command(ScrapyCommand):
|
||||
|
||||
requires_project = False
|
||||
|
|
@ -59,7 +68,8 @@ class Command(ScrapyCommand):
|
|||
if len(args) != 2:
|
||||
raise UsageError()
|
||||
|
||||
name, domain = args[0:2]
|
||||
name, url = args[0:2]
|
||||
domain = extract_domain(url)
|
||||
module = sanitize_module_name(name)
|
||||
|
||||
if self.settings.get('BOT_NAME') == module:
|
||||
|
|
|
|||
|
|
@ -3,6 +3,7 @@ import json
|
|||
import optparse
|
||||
import os
|
||||
import platform
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
|
|
@ -94,6 +95,15 @@ class ProjectTest(unittest.TestCase):
|
|||
|
||||
return p, to_unicode(stdout), to_unicode(stderr)
|
||||
|
||||
def find_in_file(self, filename, regex):
|
||||
"""Find first pattern occurrence in file"""
|
||||
pattern = re.compile(regex)
|
||||
with open(filename, "r") as f:
|
||||
for line in f:
|
||||
match = pattern.search(line)
|
||||
if match is not None:
|
||||
return match
|
||||
|
||||
|
||||
class StartprojectTest(ProjectTest):
|
||||
|
||||
|
|
@ -482,6 +492,26 @@ class GenspiderCommandTest(CommandTest):
|
|||
def test_same_filename_as_existing_spider_force(self):
|
||||
self.test_same_filename_as_existing_spider(force=True)
|
||||
|
||||
def test_url(self, url='test.com', domain="test.com"):
|
||||
self.assertEqual(0, self.call('genspider', '--force', 'test_name', url))
|
||||
self.assertEqual(domain,
|
||||
self.find_in_file(join(self.proj_mod_path,
|
||||
'spiders', 'test_name.py'),
|
||||
r'allowed_domains\s*=\s*\[\'(.+)\'\]').group(1))
|
||||
self.assertEqual('http://%s/' % domain,
|
||||
self.find_in_file(join(self.proj_mod_path,
|
||||
'spiders', 'test_name.py'),
|
||||
r'start_urls\s*=\s*\[\'(.+)\'\]').group(1))
|
||||
|
||||
def test_url_schema(self):
|
||||
self.test_url('http://test.com', 'test.com')
|
||||
|
||||
def test_url_path(self):
|
||||
self.test_url('test.com/some/other/page', 'test.com')
|
||||
|
||||
def test_url_schema_path(self):
|
||||
self.test_url('https://test.com/some/other/page', 'test.com')
|
||||
|
||||
|
||||
class GenspiderStandaloneCommandTest(ProjectTest):
|
||||
|
||||
|
|
|
|||
Loading…
Reference in New Issue