diff --git a/scrapy/trunk/scrapy/contrib/rulengine/__init__.py b/scrapy/trunk/scrapy/contrib/rulengine/__init__.py new file mode 100644 index 000000000..9486a8ab9 --- /dev/null +++ b/scrapy/trunk/scrapy/contrib/rulengine/__init__.py @@ -0,0 +1,74 @@ +""" +Here put the rules that the rules engine / pipline will take to process it. +""" +from scrapy.core.exceptions import NotConfigured +from scrapy.utils.misc import load_class +from scrapy.core import log +from scrapy.conf import settings + +class Rule(object): + """ + Interface of the Rules. + Implement this class to create a new rule. + """ + def __init__(self, wresponse=None): + self.__responsewrapper = wresponse + + def __getresponsewrapper(self): + return self.__responsewrapper + def __setresponsewrapper(self, wresponse): + self.__responsewrapper = wresponse + + responsewrapper = property(__getresponsewrapper, __setresponsewrapper) + + def check(self): + result = 0.0 + if self.responsewrapper: + result = self.holds() + if result < 0 or result > 1: + raise ValueError, "Value must be between 0 and 1." + return result + + def holds(self): + """ + User of this class must override this method. + Put here the conditions that must be satisfied by the rule. + The return value must be a number between 0.0 and 1.0. + """ + return 0.0 + +class RulesManager(object): + """ + This class contains the RulesManager which takes care of loading and + keeping track of all enabled rules. It also contains an instantiated + RulesManager (rules) to be used as singleton. + The RulesManager contains the rules classes, not instances of the rules + classes, this approach give us more flexiblility in our Rules Engine. + """ + + def __init__(self): + self.loaded = False + self.enabled = {} + + def load(self): + """ + Load enabled extensions in settings module + """ + + self.loaded = False + self.enabled.clear() + + for extension_path in settings.getlist('SIMPAGES_RULES'): + cls = load_class(extension_path) + self.enabled[cls.__name__] = cls + + self.loaded = True + + def reload(self): + self.load() + +rules = RulesManager() + + + + diff --git a/scrapy/trunk/scrapy/contrib/rulengine/exceptions.py b/scrapy/trunk/scrapy/contrib/rulengine/exceptions.py new file mode 100644 index 000000000..fa2756bc1 --- /dev/null +++ b/scrapy/trunk/scrapy/contrib/rulengine/exceptions.py @@ -0,0 +1,9 @@ +""" +Contains exceptions of the simpages stuffs +""" + +class RulesNotLoaded(Exception): + """ + Indicate that the rules was not loaded. + """ + pass diff --git a/scrapy/trunk/scrapy/contrib/rulengine/pipeline.py b/scrapy/trunk/scrapy/contrib/rulengine/pipeline.py new file mode 100644 index 000000000..227a2f792 --- /dev/null +++ b/scrapy/trunk/scrapy/contrib/rulengine/pipeline.py @@ -0,0 +1,54 @@ +""" +Represent the pipepline of execution to the rules engine. +""" +from scrapy.core import log +from scrapy.contrib.rulengine.exceptions import RulesNotLoaded + +from scrapy.contrib.rulengine import rules + +class RulesPipeline(object): + + rulesLoaded = [] + loaded = False + + def __init__(self, wresponse): + """ + wresponse: is a response wrapper object that contain the response. + """ + self.__rules = [] + self._responsewrapper = wresponse + + @staticmethod + def loadRules(): + RulesPipeline.loaded = True + rules.load() + try: + for rulename in rules.enabled.keys(): + ldr_msg = 'Loading ... %s' % rulename + ruleClass = rules.enabled[rulename] + RulesPipeline.rulesLoaded.append(ruleClass()) + log.msg(ldr_msg) + print ldr_msg + except Exception, e: + RulesPipeline.loaded = False + RulesPipeline.rulesLoaded = [] + log.msg(e) + print e + + def execute(self): + """ + Return a dictionary that conatins all the rules executed. + """ + if RulesPipeline.loaded: + rules_loaded = RulesPipeline.rulesLoaded + info_dict = {} + info_dict['rules_executed'] = {} + total = 0.0 + for rule in rules_loaded: + rule.responsewrapper = self._responsewrapper + rule_result = rule.check() + total += rule_result + info_dict['rules_executed'][rule.__class__.__name__] = rule_result + return info_dict + else: + raise RulesNotLoaded, 'Problems loading the rules.' diff --git a/scrapy/trunk/scrapy/contrib/rulengine/responseWrapper.py b/scrapy/trunk/scrapy/contrib/rulengine/responseWrapper.py new file mode 100644 index 000000000..ac66e902c --- /dev/null +++ b/scrapy/trunk/scrapy/contrib/rulengine/responseWrapper.py @@ -0,0 +1,59 @@ +""" +This class serve as wrapper of the response object. +The object created by this class are not response, but are objects that contains a response a a lot of useful information about the response. +""" + +import re + +from BeautifulSoup import Tag + +def extractText(soup): + text_in_tags = soup.findAll(text=True) + res = [] + for tag in text_in_tags: + if isinstance(tag.parent, Tag) and tag.parent.name not in('script'): + res.append(tag) + return res + +class ResponseWrapper(object): + def __init__(self, response): + self._response = response + self.__soup_bodytext = extractText(response.soup.body) if response else None + self.__soup_headtext = extractText(response.soup.head) if response else None + self._soup = None + self._bodytext = None + self._headtext = None + self._cleanbodytext = None + + def __cleantext(self, souptext): + return filter(lambda x:re.search('\w+', x), souptext) + + @property + def soup(self): + if (self._soup): + self._soup = self._response.soup + return self._response.soup + + @property + def bodytext(self): + if not self._bodytext: + self._bodytext = self.__cleantext(self.__soup_bodytext) + return self._bodytext + + @property + def headtext(self): + if not self._headtext: + self._headtext = self.__cleantext(self.__soup_headtext) + return self._headtext + + @property + def cleanbodytext(self): + if not self._cleanbodytext: + text = ' '.join(self.bodytext) + text = text.lower() + text = text.replace('\n', '') + text = text.replace('\t', '') + text = re.sub('&.*?;', '', text) + self._cleanbodytext = text + return self._cleanbodytext +