Extract domain from genspider URL (#4439)

This commit is contained in:
Matsievskiy S.V 2021-08-24 12:05:50 +03:00 committed by GitHub
parent ada539a63a
commit 3f635eb683
No known key found for this signature in database
GPG Key ID: 4AEE18F83AFDEB23
2 changed files with 41 additions and 1 deletions

View File

@ -4,6 +4,7 @@ import string
from importlib import import_module
from os.path import join, dirname, abspath, exists, splitext
from urllib.parse import urlparse
import scrapy
from scrapy.commands import ScrapyCommand
@ -22,6 +23,14 @@ def sanitize_module_name(module_name):
return module_name
def extract_domain(url):
"""Extract domain name from URL string"""
o = urlparse(url)
if o.scheme == '' and o.netloc == '':
o = urlparse("//" + url.lstrip("/"))
return o.netloc
class Command(ScrapyCommand):
requires_project = False
@ -59,7 +68,8 @@ class Command(ScrapyCommand):
if len(args) != 2:
raise UsageError()
name, domain = args[0:2]
name, url = args[0:2]
domain = extract_domain(url)
module = sanitize_module_name(name)
if self.settings.get('BOT_NAME') == module:

View File

@ -3,6 +3,7 @@ import json
import optparse
import os
import platform
import re
import subprocess
import sys
import tempfile
@ -94,6 +95,15 @@ class ProjectTest(unittest.TestCase):
return p, to_unicode(stdout), to_unicode(stderr)
def find_in_file(self, filename, regex):
"""Find first pattern occurrence in file"""
pattern = re.compile(regex)
with open(filename, "r") as f:
for line in f:
match = pattern.search(line)
if match is not None:
return match
class StartprojectTest(ProjectTest):
@ -482,6 +492,26 @@ class GenspiderCommandTest(CommandTest):
def test_same_filename_as_existing_spider_force(self):
self.test_same_filename_as_existing_spider(force=True)
def test_url(self, url='test.com', domain="test.com"):
self.assertEqual(0, self.call('genspider', '--force', 'test_name', url))
self.assertEqual(domain,
self.find_in_file(join(self.proj_mod_path,
'spiders', 'test_name.py'),
r'allowed_domains\s*=\s*\[\'(.+)\'\]').group(1))
self.assertEqual('http://%s/' % domain,
self.find_in_file(join(self.proj_mod_path,
'spiders', 'test_name.py'),
r'start_urls\s*=\s*\[\'(.+)\'\]').group(1))
def test_url_schema(self):
self.test_url('http://test.com', 'test.com')
def test_url_path(self):
self.test_url('test.com/some/other/page', 'test.com')
def test_url_schema_path(self):
self.test_url('https://test.com/some/other/page', 'test.com')
class GenspiderStandaloneCommandTest(ProjectTest):