diff --git a/scrapy/trunk/scrapy/contrib/downloadermiddleware/cache.py b/scrapy/trunk/scrapy/contrib/downloadermiddleware/cache.py index f809a9f50..ea24608cb 100644 --- a/scrapy/trunk/scrapy/contrib/downloadermiddleware/cache.py +++ b/scrapy/trunk/scrapy/contrib/downloadermiddleware/cache.py @@ -160,7 +160,7 @@ class Cache(object): headers = Headers(responseheaders) status = metadata['status'] - response = CachedResponse(domain=domain, url=url, headers=headers, status=status, body=responsebody) + response = CachedResponse(url=url, headers=headers, status=status, body=responsebody) return response def store(self, domain, key, request, response): diff --git a/scrapy/trunk/scrapy/core/downloader/handlers.py b/scrapy/trunk/scrapy/core/downloader/handlers.py index 59bb137a6..d4c43099b 100644 --- a/scrapy/trunk/scrapy/core/downloader/handlers.py +++ b/scrapy/trunk/scrapy/core/downloader/handlers.py @@ -49,7 +49,7 @@ def download_http(request, spider): body = body or '' status = int(factory.status) headers = Headers(factory.response_headers) - r = Response(domain=spider.domain_name, url=request.url, status=status, headers=headers, body=body) + r = Response(url=request.url, status=status, headers=headers, body=body) signals.send_catch_log(signal=signals.request_uploaded, sender='download_http', request=request, spider=spider) signals.send_catch_log(signal=signals.response_downloaded, sender='download_http', response=r, spider=spider) return r @@ -81,5 +81,5 @@ def download_file(request, spider) : """Return a deferred for a file download.""" filepath = request.url.split("file://")[1] with open(filepath) as f: - response = Response(domain=spider.domain_name, url=request.url, body=f.read()) + response = Response(url=request.url, body=f.read()) return defer_succeed(response) diff --git a/scrapy/trunk/scrapy/http/response.py b/scrapy/trunk/scrapy/http/response.py index 3ea480611..38318caff 100644 --- a/scrapy/trunk/scrapy/http/response.py +++ b/scrapy/trunk/scrapy/http/response.py @@ -7,7 +7,6 @@ See documentation in docs/ref/request-response.rst import re import copy -from types import NoneType from twisted.web.http import RESPONSES from BeautifulSoup import UnicodeDammit @@ -19,8 +18,7 @@ class Response(object): _ENCODING_RE = re.compile(r'charset=([\w-]+)', re.I) - def __init__(self, domain, url, status=200, headers=None, body=None, meta=None): - self.domain = domain + def __init__(self, url, status=200, headers=None, body=None, meta=None): self.url = Url(url) self.headers = Headers(headers or {}) self.status = status @@ -43,8 +41,8 @@ class Response(object): return encoding.group(1) def __repr__(self): - return "Response(domain=%s, url=%s, headers=%s, status=%s, body=%s)" % \ - (repr(self.domain), repr(self.url), repr(self.headers), repr(self.status), repr(self.body)) + return "Response(url=%s, headers=%s, status=%s, body=%s)" % \ + (repr(self.url), repr(self.headers), repr(self.status), repr(self.body)) def __str__(self): if self.status == 200: @@ -56,7 +54,7 @@ class Response(object): """Create a new Response based on the current one""" return self.replace() - def replace(self, domain=None, url=None, status=None, headers=None, body=None): + def replace(self, url=None, status=None, headers=None, body=None): """Create a new Response with the same attributes except for those given new values. @@ -64,8 +62,7 @@ class Response(object): >>> newresp = oldresp.replace(body="New body") """ - new = self.__class__(domain=domain or self.domain, - url=url or self.url, + new = self.__class__(url=url or self.url, status=status or self.status, headers=headers or copy.deepcopy(self.headers), body=body) diff --git a/scrapy/trunk/scrapy/tests/test_adaptors.py b/scrapy/trunk/scrapy/tests/test_adaptors.py index 7d1543db7..af5f00ed7 100644 --- a/scrapy/trunk/scrapy/tests/test_adaptors.py +++ b/scrapy/trunk/scrapy/tests/test_adaptors.py @@ -42,7 +42,7 @@ class AdaptorsTestCase(unittest.TestCase): def get_selector(self, domain, url, sample_filename, headers=None, selector=HtmlXPathSelector): sample_filename = os.path.join(self.samplesdir, sample_filename) body = file(sample_filename).read() - response = Response(domain=domain, url=url, headers=Headers(headers), status=200, body=body) + response = Response(url=url, headers=Headers(headers), status=200, body=body) return selector(response) def test_extract(self): @@ -124,7 +124,7 @@ class AdaptorsTestCase(unittest.TestCase): something2 """ - sample_response = Response('foobar.com', 'http://foobar.com/dummy', body=test_data) + sample_response = Response('http://foobar.com/dummy', body=test_data) sample_xsel = XmlXPathSelector(sample_response) sample_adaptor = adaptors.ExtractImages(response=sample_response) diff --git a/scrapy/trunk/scrapy/tests/test_contrib_response_soup.py b/scrapy/trunk/scrapy/tests/test_contrib_response_soup.py index fee8e77e7..82c291038 100644 --- a/scrapy/trunk/scrapy/tests/test_contrib_response_soup.py +++ b/scrapy/trunk/scrapy/tests/test_contrib_response_soup.py @@ -11,7 +11,7 @@ class ResponseSoupTest(unittest.TestCase): ResponseSoup() def test_response_soup(self): - r1 = Response('example.com', 'http://www.example.com', body='') + r1 = Response('http://www.example.com', body='') soup1 = r1.getsoup() soup2 = r1.getsoup() @@ -22,7 +22,7 @@ class ResponseSoupTest(unittest.TestCase): assert soup1 is soup2 def test_response_soup_caching(self): - r1 = Response('example.com', 'http://www.example.com', body='') + r1 = Response('http://www.example.com', body='') soup1 = r1.getsoup() r2 = r1.copy() soup2 = r1.getsoup() diff --git a/scrapy/trunk/scrapy/tests/test_engine.py b/scrapy/trunk/scrapy/tests/test_engine.py index 2ae84d9d9..54d2771a0 100644 --- a/scrapy/trunk/scrapy/tests/test_engine.py +++ b/scrapy/trunk/scrapy/tests/test_engine.py @@ -152,7 +152,6 @@ class EngineTest(unittest.TestCase): self.assertEqual(404, response.status) if session.getpath(response.url) == '/redirect': self.assertEqual(302, response.status) - self.assertEqual(response.domain, spider.domain_name) def test_item_data(self): """ diff --git a/scrapy/trunk/scrapy/tests/test_http_request.py b/scrapy/trunk/scrapy/tests/test_http_request.py index cee41cb15..cda8423a1 100644 --- a/scrapy/trunk/scrapy/tests/test_http_request.py +++ b/scrapy/trunk/scrapy/tests/test_http_request.py @@ -5,6 +5,9 @@ from scrapy.core.scheduler import GroupFilter class RequestTest(unittest.TestCase): def test_init(self): + # Request requires url in the constructor + self.assertRaises(Exception, Request) + r = Request("http://www.example.com") assert isinstance(r.url, Url) self.assertEqual(r.url, "http://www.example.com") diff --git a/scrapy/trunk/scrapy/tests/test_http_response.py b/scrapy/trunk/scrapy/tests/test_http_response.py index 117bd8ab0..4877f1ea9 100644 --- a/scrapy/trunk/scrapy/tests/test_http_response.py +++ b/scrapy/trunk/scrapy/tests/test_http_response.py @@ -5,17 +5,16 @@ from scrapy.http.response import _ResponseBody class ResponseTest(unittest.TestCase): def test_init(self): - # Response requires domain and url + # Response requires url in the consturctor self.assertRaises(Exception, Response) - self.assertRaises(Exception, Response, 'example.com') - self.assertTrue(isinstance(Response('example.com', 'http://example.com/'), Response)) + self.assertTrue(isinstance(Response('http://example.com/'), Response)) # body can be str or None but not ResponseBody - self.assertTrue(isinstance(Response('example.com', 'http://example.com/', body=None), Response)) - self.assertTrue(isinstance(Response('example.com', 'http://example.com/', body='body'), Response)) + self.assertTrue(isinstance(Response('http://example.com/', body=None), Response)) + self.assertTrue(isinstance(Response('http://example.com/', body='body'), Response)) # test presence of all optional parameters - self.assertTrue(isinstance(Response('example.com', 'http://example.com/', headers={}, status=200, body=None), Response)) + self.assertTrue(isinstance(Response('http://example.com/', headers={}, status=200, body=None), Response)) - r = Response("domain.com", "http://www.example.com") + r = Response("http://www.example.com") assert isinstance(r.url, Url) self.assertEqual(r.url, "http://www.example.com") self.assertEqual(r.status, 200) @@ -27,7 +26,7 @@ class ResponseTest(unittest.TestCase): meta = {"lala": "lolo"} headers = {"caca": "coco"} body = "a body" - r = Response("example.com", "http://www.example.com", meta=meta, headers=headers, body="a body") + r = Response("http://www.example.com", meta=meta, headers=headers, body="a body") assert r.meta is not meta self.assertEqual(r.meta, meta) @@ -37,7 +36,7 @@ class ResponseTest(unittest.TestCase): def test_copy(self): """Test Response copy""" - r1 = Response('example.com', "http://www.example.com") + r1 = Response("http://www.example.com") r1.meta['foo'] = 'bar' r1.cache['lala'] = 'lolo' r2 = r1.copy() @@ -55,16 +54,16 @@ class ResponseTest(unittest.TestCase): class CustomResponse(Response): pass - r1 = CustomResponse('example.com', 'http://www.example.com') + r1 = CustomResponse('http://www.example.com') r2 = r1.copy() assert type(r2) is CustomResponse def test_httprepr(self): - r1 = Response('example.com', "http://www.example.com") + r1 = Response("http://www.example.com") self.assertEqual(r1.httprepr(), 'HTTP/1.1 200 OK\r\n\r\n') - r1 = Response('example.com', "http://www.example.com", status=404, headers={"Content-type": "text/html"}, body="Some body") + r1 = Response("http://www.example.com", status=404, headers={"Content-type": "text/html"}, body="Some body") self.assertEqual(r1.httprepr(), 'HTTP/1.1 404 Not Found\r\nContent-Type: text/html\r\n\r\nSome body\r\n') class ResponseBodyTest(unittest.TestCase): diff --git a/scrapy/trunk/scrapy/tests/test_libxml2.py b/scrapy/trunk/scrapy/tests/test_libxml2.py index 0d256742d..af363342f 100644 --- a/scrapy/trunk/scrapy/tests/test_libxml2.py +++ b/scrapy/trunk/scrapy/tests/test_libxml2.py @@ -35,7 +35,7 @@ class ResponseLibxml2DocTest(TestCase): scrapymanager.configure() self.body_content = 'test problematic \x00 body' - response = Response('example.com', 'http://example.com/catalog/product/blabla-123', + response = Response('http://example.com/catalog/product/blabla-123', headers={'Content-Type': 'text/plain; charset=utf-8'}, body=self.body_content) response.getlibxml2doc() diff --git a/scrapy/trunk/scrapy/tests/test_link.py b/scrapy/trunk/scrapy/tests/test_link.py index ed09f3b7b..e7be4f6f7 100644 --- a/scrapy/trunk/scrapy/tests/test_link.py +++ b/scrapy/trunk/scrapy/tests/test_link.py @@ -15,7 +15,7 @@ class LinkExtractorTestCase(unittest.TestCase):

Other category

""" - response = Response("example.org", "http://example.org/somepage/index.html", body=html) + response = Response("http://example.org/somepage/index.html", body=html) lx = LinkExtractor() # default: tag=a, attr=href self.assertEqual(lx.extract_links(response), @@ -28,7 +28,7 @@ class LinkExtractorTestCase(unittest.TestCase): html = """Page title<title><base href="http://otherdomain.com/base/" /> <body><p><a href="item/12.html">Item 12</a></p> </body></html>""" - response = Response("example.org", "http://example.org/somepage/index.html", body=html) + response = Response("http://example.org/somepage/index.html", body=html) lx = LinkExtractor() # default: tag=a, attr=href self.assertEqual(lx.extract_links(response), @@ -37,10 +37,10 @@ class LinkExtractorTestCase(unittest.TestCase): def test_extraction_encoding(self): base_path = os.path.join(os.path.dirname(__file__), 'sample_data', 'link_extractor') body = open(os.path.join(base_path, 'linkextractor_noenc.html'), 'r').read() - response_utf8 = Response(url='http://example.com/utf8', domain='example.com', body=body, headers={'Content-Type': ['text/html; charset=utf-8']}) - response_noenc = Response(url='http://example.com/noenc', domain='example.com', body=body) + response_utf8 = Response(url='http://example.com/utf8', body=body, headers={'Content-Type': ['text/html; charset=utf-8']}) + response_noenc = Response(url='http://example.com/noenc', body=body) body = open(os.path.join(base_path, 'linkextractor_latin1.html'), 'r').read() - response_latin1 = Response(url='http://example.com/latin1', domain='example.com', body=body) + response_latin1 = Response(url='http://example.com/latin1', body=body) lx = LinkExtractor() self.assertEqual(lx.extract_links(response_utf8), @@ -67,7 +67,7 @@ class RegexLinkExtractorTestCase(unittest.TestCase): def setUp(self): base_path = os.path.join(os.path.dirname(__file__), 'sample_data', 'link_extractor') body = open(os.path.join(base_path, 'regex_linkextractor.html'), 'r').read() - self.response = Response(url='http://example.com/index', domain='example.com', body=body) + self.response = Response(url='http://example.com/index', body=body) def test_urls_type(self): '''Test that the resulting urls are regular strings and not a unicode objects''' @@ -146,7 +146,7 @@ class RegexLinkExtractorTestCase(unittest.TestCase): # def setUp(self): # base_path = os.path.join(os.path.dirname(__file__), 'sample_data', 'link_extractor') # body = open(os.path.join(base_path 'image_linkextractor.html'), 'r').read() -# self.response = Response(url='http://example.com/index', domain='example.com', body=body) +# self.response = Response(url='http://example.com/index', body=body) # def test_urls_type(self): # '''Test that the resulting urls are regular strings and not a unicode objects''' diff --git a/scrapy/trunk/scrapy/tests/test_middleware_decompression.py b/scrapy/trunk/scrapy/tests/test_middleware_decompression.py index 0a86b84b3..1ec5b786a 100644 --- a/scrapy/trunk/scrapy/tests/test_middleware_decompression.py +++ b/scrapy/trunk/scrapy/tests/test_middleware_decompression.py @@ -16,7 +16,7 @@ def setUp(): fd = open(os.path.join(datadir, 'feed-sample1.' + format), 'r') body = fd.read() fd.close() - test_responses[format] = Response('foo.com', 'http://foo.com/bar', body=body) + test_responses[format] = Response('http://foo.com/bar', body=body) return uncompressed_body, test_responses class ScrapyDecompressionTest(TestCase): diff --git a/scrapy/trunk/scrapy/tests/test_middleware_retry.py b/scrapy/trunk/scrapy/tests/test_middleware_retry.py index b028508c2..7f6024bff 100644 --- a/scrapy/trunk/scrapy/tests/test_middleware_retry.py +++ b/scrapy/trunk/scrapy/tests/test_middleware_retry.py @@ -12,8 +12,8 @@ class RetryTest(unittest.TestCase): self.spider = spiders.fromdomain('scrapytest.org') def test_process_exception(self): - exception_404 = (Request('http://www.scrapytest.org/404'), HttpException('404', None, Response('scrapytest.org', 'http://www.scrapytest.org/404', body='')), self.spider) - exception_503 = (Request('http://www.scrapytest.org/503'), HttpException('503', None, Response('scrapytest.org', 'http://www.scrapytest.org/503', body='')), self.spider) + exception_404 = (Request('http://www.scrapytest.org/404'), HttpException('404', None, Response('http://www.scrapytest.org/404', body='')), self.spider) + exception_503 = (Request('http://www.scrapytest.org/503'), HttpException('503', None, Response('http://www.scrapytest.org/503', body='')), self.spider) mw = RetryMiddleware() mw.retry_times = 1 diff --git a/scrapy/trunk/scrapy/tests/test_utils_iterators.py b/scrapy/trunk/scrapy/tests/test_utils_iterators.py index e9adaad40..70cf95028 100644 --- a/scrapy/trunk/scrapy/tests/test_utils_iterators.py +++ b/scrapy/trunk/scrapy/tests/test_utils_iterators.py @@ -20,7 +20,7 @@ class UtilsXmlTestCase(unittest.TestCase): </product>\ </products>""" - response = Response(domain="example.com", url="http://example.com", body=body) + response = Response(url="http://example.com", body=body) attrs = [] for x in xmliter(response, 'product'): attrs.append((x.x("@id").extract(), x.x("name/text()").extract(), x.x("./type/text()").extract())) @@ -53,7 +53,7 @@ class UtilsXmlTestCase(unittest.TestCase): </channel> </rss> """ - response = Response(domain='mydummycompany.com', url='http://mydummycompany.com', body=body) + response = Response(url='http://mydummycompany.com', body=body) my_iter = xmliter(response, 'item') node = my_iter.next() @@ -83,7 +83,7 @@ class UtilsCsvTestCase(unittest.TestCase): def test_iterator_defaults(self): body = open(self.sample_feed_path).read() - response = Response(domain="example.com", url="http://example.com/", body=body) + response = Response(url="http://example.com/", body=body) csv = csviter(response) result = [row for row in csv] @@ -101,7 +101,7 @@ class UtilsCsvTestCase(unittest.TestCase): def test_iterator_delimiter(self): body = open(self.sample_feed_path).read().replace(',', '\t') - response = Response(domain="example.com", url="http://example.com/", body=body) + response = Response(url="http://example.com/", body=body) csv = csviter(response, delimiter='\t') self.assertEqual([row for row in csv], @@ -114,7 +114,7 @@ class UtilsCsvTestCase(unittest.TestCase): sample = open(self.sample_feed_path).read().splitlines() headers, body = sample[0].split(','), '\n'.join(sample[1:]) - response = Response(domain="example.com", url="http://example.com/", body=body) + response = Response(url="http://example.com/", body=body) csv = csviter(response, headers=headers) self.assertEqual([row for row in csv], @@ -127,7 +127,7 @@ class UtilsCsvTestCase(unittest.TestCase): body = open(self.sample_feed_path).read() body = '\n'.join((body, 'a,b', 'a,b,c,d')) - response = Response(domain="example.com", url="http://example.com/", body=body) + response = Response(url="http://example.com/", body=body) csv = csviter(response) self.assertEqual([row for row in csv], @@ -139,7 +139,7 @@ class UtilsCsvTestCase(unittest.TestCase): def test_iterator_exception(self): body = open(self.sample_feed_path).read() - response = Response(domain="example.com", url="http://example.com/", body=body) + response = Response(url="http://example.com/", body=body) iter = csviter(response) iter.next() iter.next() diff --git a/scrapy/trunk/scrapy/tests/test_utils_response.py b/scrapy/trunk/scrapy/tests/test_utils_response.py index feec740ad..2e86baf1e 100644 --- a/scrapy/trunk/scrapy/tests/test_utils_response.py +++ b/scrapy/trunk/scrapy/tests/test_utils_response.py @@ -3,7 +3,7 @@ from scrapy.http.response import Response from scrapy.utils.response import body_or_str class ResponseUtilsTest(unittest.TestCase): - dummy_response = Response(domain='example.org', url='http://example.org/', body='dummy_response') + dummy_response = Response(url='http://example.org/', body='dummy_response') def test_input(self): self.assertTrue(isinstance(body_or_str(self.dummy_response), basestring)) diff --git a/scrapy/trunk/scrapy/tests/test_xpath.py b/scrapy/trunk/scrapy/tests/test_xpath.py index b046605a3..8d5afa77c 100644 --- a/scrapy/trunk/scrapy/tests/test_xpath.py +++ b/scrapy/trunk/scrapy/tests/test_xpath.py @@ -19,7 +19,7 @@ class XPathTestCase(unittest.TestCase): def test_selector_simple(self): """Simple selector tests""" body = "<p><input name='a'value='1'/><input name='b'value='2'/></p>" - response = Response(domain="example.com", url="http://example.com", body=body) + response = Response(url="http://example.com", body=body) xpath = HtmlXPathSelector(response) xl = xpath.x('//input') @@ -75,7 +75,7 @@ class XPathTestCase(unittest.TestCase): </div> </body>""" - response = Response(domain="example.com", url="http://example.com", body=body) + response = Response(url="http://example.com", body=body) x = HtmlXPathSelector(response) divtwo = x.x('//div[@class="two"]') @@ -100,7 +100,7 @@ class XPathTestCase(unittest.TestCase): </div> """ - response = Response(domain="example.com", url="http://example.com", body=body) + response = Response(url="http://example.com", body=body) x = HtmlXPathSelector(response) name_re = re.compile("Name: (\w+)") @@ -131,7 +131,7 @@ class XPathTestCase(unittest.TestCase): </test> """ - response = Response(domain="example.com", url="http://example.com", body=body) + response = Response(url="http://example.com", body=body) x = XmlXPathSelector(response) x.register_namespace("somens", "http://scrapy.org") @@ -149,7 +149,7 @@ class XPathTestCase(unittest.TestCase): <p:SecondTestTag><material/><price>90</price><p:name>Dried Rose</p:name></p:SecondTestTag> </BrowseNode> """ - response = Response(domain="example.com", url="http://example.com", body=body) + response = Response(url="http://example.com", body=body) x = XmlXPathSelector(response) x.register_namespace("xmlns", "http://webservices.amazon.com/AWSECommerceService/2005-10-05") @@ -176,7 +176,7 @@ class XPathTestCase(unittest.TestCase): html_utf8 = html.encode(encoding) headers = {'Content-Type': ['text/html; charset=utf-8']} - response = Response(domain="example.com", url="http://example.com", headers=headers, body=html_utf8) + response = Response(url="http://example.com", headers=headers, body=html_utf8) x = HtmlXPathSelector(response) self.assertEquals(x.x("//span[@id='blank']/text()").extract(), [u'\xa3']) diff --git a/scrapy/trunk/scrapy/tests/test_xpath_extension.py b/scrapy/trunk/scrapy/tests/test_xpath_extension.py index b0abcb4c0..8880dcbc7 100644 --- a/scrapy/trunk/scrapy/tests/test_xpath_extension.py +++ b/scrapy/trunk/scrapy/tests/test_xpath_extension.py @@ -9,7 +9,7 @@ class ResponseLibxml2Test(unittest.TestCase): ResponseLibxml2() def test_response_libxml2_caching(self): - r1 = Response('example.com', 'http://www.example.com', body='<html><head></head><body></body></html>') + r1 = Response('http://www.example.com', body='<html><head></head><body></body></html>') r2 = r1.copy() doc1 = r1.getlibxml2doc() diff --git a/scrapy/trunk/scrapy/xpath/selector.py b/scrapy/trunk/scrapy/xpath/selector.py index da939bb91..fd8f89a21 100644 --- a/scrapy/trunk/scrapy/xpath/selector.py +++ b/scrapy/trunk/scrapy/xpath/selector.py @@ -28,7 +28,7 @@ class XPathSelector(object): self.doc = Libxml2Document(response, constructor=constructor) self.xmlNode = self.doc.xmlDoc elif text: - response = Response(domain=None, url=None, body=unicode_to_str(text)) + response = Response(url=None, body=unicode_to_str(text)) self.doc = Libxml2Document(response, constructor=constructor) self.xmlNode = self.doc.xmlDoc self.expr = expr