From 127e5ae02932af67c6157939cff6ab388c89c677 Mon Sep 17 00:00:00 2001 From: Ismael Carnales Date: Mon, 2 Mar 2009 18:03:48 +0000 Subject: [PATCH] convert process_attr to a parameter in contructor so extending the class is not needed --HG-- extra : convert_revision : svn%3Ab85faa78-f9eb-468e-a121-7cced6da292c%40954 --- scrapy/trunk/scrapy/contrib_exp/link/__init__.py | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/scrapy/trunk/scrapy/contrib_exp/link/__init__.py b/scrapy/trunk/scrapy/contrib_exp/link/__init__.py index 54ac0117b..41efb4420 100644 --- a/scrapy/trunk/scrapy/contrib_exp/link/__init__.py +++ b/scrapy/trunk/scrapy/contrib_exp/link/__init__.py @@ -28,15 +28,18 @@ class LinkExtractor(HTMLParser): * attr (string or function) * an attribute name which is used to search for links (defaults to "href") * a function which receives an attribute name and returns whether to scan it + * process (funtion) + * a function wich receives the attribute value before assigning it * unique - if True the same urls won't be extracted twice, otherwise the same urls will be extracted multiple times (with potentially different link texts) """ - def __init__(self, tag="a", attr="href", unique=False): + def __init__(self, tag="a", attr="href", process=None, unique=False): HTMLParser.__init__(self) self.scan_tag = tag if callable(tag) else lambda t: t == tag self.scan_attr = attr if callable(attr) else lambda a: a == attr + self.process_attr = process if callable(process) else lambda v: v self.unique = unique def _extract_links(self, response_text, response_url, response_encoding): @@ -86,11 +89,6 @@ class LinkExtractor(HTMLParser): if self.current_link and not self.current_link.text: self.current_link.text = data.strip() - def process_attr(self, value): - """Hook to process the value of the attribute before asigning - it to the link""" - return value - def matches(self, url): """This extractor matches with any url, since it doesn't contain any patterns"""