mirror of https://github.com/scrapy/scrapy.git
Modified image pipeline to make it able to manage 304 responses
--HG-- extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40385
This commit is contained in:
parent
9a7b5a7cbf
commit
d5469eabb2
|
|
@ -4,6 +4,7 @@ import time
|
|||
import hashlib
|
||||
import urllib
|
||||
import urlparse
|
||||
from datetime import datetime
|
||||
from cStringIO import StringIO
|
||||
|
||||
import Image
|
||||
|
|
@ -17,7 +18,7 @@ from scrapy.conf import settings
|
|||
from scrapy.contrib.pipeline.media import MediaPipeline
|
||||
|
||||
# the age at which we download images again
|
||||
IMAGE_EXPIRES = settings.getint('IMAGES_EXPIRES', 90)
|
||||
IMAGE_EXPIRES = settings.getint('IMAGES_EXPIRES', 15)
|
||||
|
||||
class NoimagesDrop(DropItem):
|
||||
pass
|
||||
|
|
@ -50,12 +51,10 @@ class ImagesPipeline(MediaPipeline):
|
|||
|
||||
def media_to_download(self, request, info):
|
||||
relative, absolute = self._get_paths(request)
|
||||
if not should_download(absolute):
|
||||
self.inc_stats(info.domain, 'uptodate')
|
||||
referer = request.headers.get('Referer')
|
||||
log.msg('Image (uptodate): Downloaded %s from %s referred in <%s>' % \
|
||||
(self.MEDIA_TYPE, request, referer), level=log.DEBUG, domain=info.domain)
|
||||
return relative
|
||||
expired, mtime = image_expired(absolute)
|
||||
if not expired:
|
||||
mod_date = datetime.utcfromtimestamp(mtime)
|
||||
request.headers['If-Modified-Since'] = mod_date.strftime('%a, %d %b %Y %H:%M:%S GMT')
|
||||
|
||||
def media_downloaded(self, response, request, info):
|
||||
mtype = self.MEDIA_TYPE
|
||||
|
|
@ -75,8 +74,15 @@ class ImagesPipeline(MediaPipeline):
|
|||
return result
|
||||
|
||||
def media_failed(self, failure, request, info):
|
||||
mtype = self.MEDIA_TYPE
|
||||
referer = request.headers.get('Referer')
|
||||
errmsg = str(failure.value) if isinstance(failure.value, HttpException) else str(failure)
|
||||
|
||||
if '304 Not Modified' in errmsg:
|
||||
msg = 'Image (notmodified): Downloaded %s from %s referred in <%s>' % (mtype, request, referer)
|
||||
log.msg(msg, level=log.DEBUG, domain=info.domain)
|
||||
return self._get_paths(request)[0]
|
||||
|
||||
msg = 'Image (http-error): Error downloading %s from %s referred in <%s>: %s' % (self.MEDIA_TYPE, request, referer, errmsg)
|
||||
log.msg(msg, level=log.WARNING, domain=info.domain)
|
||||
raise ImageException(msg)
|
||||
|
|
@ -118,16 +124,17 @@ class ImagesPipeline(MediaPipeline):
|
|||
|
||||
|
||||
|
||||
def should_download(path):
|
||||
"""Should the image downloader download the image to the location specified
|
||||
def image_expired(path):
|
||||
"""
|
||||
Is the image older than the expiration time?
|
||||
"""
|
||||
try:
|
||||
mtime = os.path.getmtime(path)
|
||||
age_seconds = time.time() - mtime
|
||||
age_days = age_seconds / 60 / 60 / 24
|
||||
return age_days > IMAGE_EXPIRES
|
||||
return (age_days > IMAGE_EXPIRES, mtime)
|
||||
except:
|
||||
return True
|
||||
return (True, 0)
|
||||
|
||||
_MULTIPLE_SLASHES_REGEXP = re.compile(r"\/{2,}")
|
||||
_FINAL_SLASH_REGEXP = re.compile(r"\/$")
|
||||
|
|
|
|||
Loading…
Reference in New Issue