diff --git a/scrapy/trunk/docs/topics/item-pipeline.rst b/scrapy/trunk/docs/topics/item-pipeline.rst index 408435b93..9b4c73084 100644 --- a/scrapy/trunk/docs/topics/item-pipeline.rst +++ b/scrapy/trunk/docs/topics/item-pipeline.rst @@ -21,12 +21,10 @@ Writing your own item pipeline Writing your own item pipeline is easy. Each item pipeline component is a single Python class that must define the following method: -.. method:: process_item(domain, response, item) +.. method:: process_item(domain, item) ``domain`` is a string with the domain of the spider which scraped the item -``response`` is a :class:`scrapy.http.Response` with the response where the item was scraped - ``item`` is a :class:`scrapy.item.ScrapedItem` with the item scraped This method is called for every item pipeline component and must either return @@ -63,7 +61,7 @@ attribute), and drops those items which don't contain a price:: vat_factor = 1.15 - def process_item(self, domain, response, item): + def process_item(self, domain, item): if item.price: if item.price_excludes_vat: item.price = item.price * self.vat_factor diff --git a/scrapy/trunk/scrapy/contrib/item/models.py b/scrapy/trunk/scrapy/contrib/item/models.py index d58c705b5..8488844fb 100644 --- a/scrapy/trunk/scrapy/contrib/item/models.py +++ b/scrapy/trunk/scrapy/contrib/item/models.py @@ -26,7 +26,7 @@ class ValidationError(DropItem): return '%s' % (self.problem) class ValidationPipeline(object): - def process_item(self, domain, response, item): + def process_item(self, domain, item): item.validate() return item diff --git a/scrapy/trunk/scrapy/contrib/itemsampler.py b/scrapy/trunk/scrapy/contrib/itemsampler.py index f31d87e02..7f190edc5 100644 --- a/scrapy/trunk/scrapy/contrib/itemsampler.py +++ b/scrapy/trunk/scrapy/contrib/itemsampler.py @@ -51,7 +51,7 @@ class ItemSamplerPipeline(object): dispatcher.connect(self.domain_closed, signal=signals.domain_closed) dispatcher.connect(self.engine_stopped, signal=signals.engine_stopped) - def process_item(self, domain, response, item): + def process_item(self, domain, item): sampled = stats.getpath("%s/items_sampled" % domain, 0) if sampled < items_per_domain: self.items[item.guid] = item diff --git a/scrapy/trunk/scrapy/contrib/pipeline/media.py b/scrapy/trunk/scrapy/contrib/pipeline/media.py index 5f19de77d..0cb9a3412 100644 --- a/scrapy/trunk/scrapy/contrib/pipeline/media.py +++ b/scrapy/trunk/scrapy/contrib/pipeline/media.py @@ -26,7 +26,7 @@ class MediaPipeline(object): def close_domain(self, domain): del self.domaininfo[domain] - def process_item(self, domain, response, item): + def process_item(self, domain, item): info = self.domaininfo[domain] requests = self.get_media_requests(item, info) assert requests is None or hasattr(requests, '__iter__'), \ diff --git a/scrapy/trunk/scrapy/contrib/pipeline/shoveitem.py b/scrapy/trunk/scrapy/contrib/pipeline/shoveitem.py index e73d43ef2..6504bc429 100644 --- a/scrapy/trunk/scrapy/contrib/pipeline/shoveitem.py +++ b/scrapy/trunk/scrapy/contrib/pipeline/shoveitem.py @@ -27,7 +27,7 @@ class ShoveItemPipeline(object): dispatcher.connect(self.domain_open, signal=signals.domain_open) dispatcher.connect(self.domain_closed, signal=signals.domain_closed) - def process_item(self, domain, response, item): + def process_item(self, domain, item): guid = str(item.guid) if guid in self.stores[domain]: diff --git a/scrapy/trunk/scrapy/contrib/pipeline/show.py b/scrapy/trunk/scrapy/contrib/pipeline/show.py index f08234be6..3e7474a49 100644 --- a/scrapy/trunk/scrapy/contrib/pipeline/show.py +++ b/scrapy/trunk/scrapy/contrib/pipeline/show.py @@ -10,6 +10,6 @@ class ShowItemPipeline(object): if not settings['DEBUG_SHOWITEM']: raise NotConfigured - def process_item(self, domain, response, item): + def process_item(self, domain, item): log.msg("Scraped: \n%s" % repr(item), log.DEBUG, domain=domain) return item diff --git a/scrapy/trunk/scrapy/core/engine.py b/scrapy/trunk/scrapy/core/engine.py index 4cd5e5c7f..970dbb687 100644 --- a/scrapy/trunk/scrapy/core/engine.py +++ b/scrapy/trunk/scrapy/core/engine.py @@ -244,7 +244,7 @@ class ExecutionEngine(object): if isinstance(item, ScrapedItem): log.msg("Scraped %s in <%s>" % (item, request.url), log.DEBUG, domain=domain) signals.send_catch_log(signal=signals.item_scraped, sender=self.__class__, item=item, spider=spider, response=response) - piped = self.pipeline.pipe(item, spider, response) # TODO: remove response + piped = self.pipeline.pipe(item, spider) piped.addBoth(_onpipelinefinish, item) elif isinstance(item, Request): signals.send_catch_log(signal=signals.request_received, sender=self.__class__, request=item, spider=spider, response=response) diff --git a/scrapy/trunk/scrapy/item/pipeline.py b/scrapy/trunk/scrapy/item/pipeline.py index 98692358b..787bc61bc 100644 --- a/scrapy/trunk/scrapy/item/pipeline.py +++ b/scrapy/trunk/scrapy/item/pipeline.py @@ -46,7 +46,7 @@ class ItemPipelineManager(object): def domain_is_idle(self, domain): return not self.domaininfo.get(domain) - def pipe(self, item, spider, response): + def pipe(self, item, spider): """ item pipelines are instanceable classes that defines a `pipeline` method that takes ScrapedItem as input and returns ScrapedItem. @@ -74,7 +74,7 @@ class ItemPipelineManager(object): current_stage = pipeline.pop(0) log.msg("_%s_ Pipeline stage: %s" % (item, type(current_stage).__name__), log.TRACE, domain=domain) - d = mustbe_deferred(current_stage.process_item, domain, response, item) + d = mustbe_deferred(current_stage.process_item, domain, item) d.addCallback(_next_stage) return d