From 609aed45539f0fb7e08fb5f39c89c827a4f1d94b Mon Sep 17 00:00:00 2001 From: Pablo Hoffman Date: Thu, 27 Aug 2009 19:33:33 -0300 Subject: [PATCH] added inspect_response() function for inspecting responses from spiders --- docs/topics/shell.rst | 73 +++++++++++++++++++++++++++++++++++++++++-- scrapy/log.py | 9 ++++++ scrapy/shell.py | 48 ++++++++++++++++++++++------ 3 files changed, 117 insertions(+), 13 deletions(-) diff --git a/docs/topics/shell.rst b/docs/topics/shell.rst index f3b40c25c..6e3de9bc1 100644 --- a/docs/topics/shell.rst +++ b/docs/topics/shell.rst @@ -53,8 +53,12 @@ Built-in Shortcuts URL and update all related objects accordingly. * ``view(response)`` - open the given response in your local web browser, for - inspection. Note that this will generate a temporary file which won't be - removed automatically. + inspection. This will add a `\ tag`_ to the response body in order + for external links (such as images and style sheets) to display properly. + Note, however,that this will create a temporary file in your computer, + which won't be removed automatically. + +.. _ tag: http://www.w3schools.com/TAGS/tag_base.asp Built-in Objects ---------------- @@ -77,7 +81,7 @@ Those objects are: fetch a new request (without leaving the shell) using the ``fetch`` shortcut. - * ``response`` - a :class:`~scrapy.http.Response` object contaning the last + * ``response`` - a :class:`~scrapy.http.Response` object containing the last fetched page * ``hxs`` - a :class:`~scrapy.selector.HtmlXPathSelector` object constructed @@ -146,3 +150,66 @@ After that, we can stary playing with the objects:: 2009-04-03 00:57:39-0300 [scrapybot] ERROR: Downloading from : 405 Method Not Allowed >>> + +Invoking the shell from spiders to inspect responses +==================================================== + +Sometimes you want to inspect the responses that are being processed in a +certain point of your spider, if only to check that response you expect is +getting there. + +This can be achieved by using the ``scrapy.shell.inspect_response`` function. + +Here's an example of how you would call it from your spider:: + + class MySpider(BaseSpider): + domain_name = 'example.com' + + def parse(self, response): + if response.url == 'http://www.example.com/products.php': + from scrapy.shell import inspect_response + inspect_response(response) + + # ... your parsing code .. + +When you the spider you will get something similar to this:: + + 2009-08-27 19:15:25-0300 [example.com] DEBUG: Crawled (referer: ) + 2009-08-27 19:15:26-0300 [example.com] DEBUG: Crawled (referer: ) + + Scrapy Shell + ============ + + Inspecting: >> response.url + 'http://www.example.com/products.php' + +Then, you can check if the extraction code is working:: + + >>> hxs.select('//h1') + [] + +Nope, it doesn't. So you can open the response in your web browser and see if +it's the response you were expecting:: + + >>> view(response) + >>> + +Finally you hit Ctrl-D (or Ctrl-Z in Windows) to exit the shell and resume the +crawling:: + + >>> ^D + 2009-08-27 19:15:25-0300 [example.com] DEBUG: Crawled (referer: ) + 2009-08-27 19:15:25-0300 [example.com] DEBUG: Crawled (referer: ) + # ... + +Note that you can't use the ``fetch`` shortcut here since the Scrapy engine is +blocked by the shell. However, after you leave the shell, the spider will +continue crawling where it stopped, as shown above. + diff --git a/scrapy/log.py b/scrapy/log.py index 4d621cb6d..d36d0cf88 100644 --- a/scrapy/log.py +++ b/scrapy/log.py @@ -34,6 +34,8 @@ log_level = DEBUG started = False +_prev_descriptors = (sys.stdout, sys.stderr) + def start(logfile=None, loglevel=None, logstdout=None): """Initialize and start logging facility""" global log_level, started @@ -53,6 +55,13 @@ def start(logfile=None, loglevel=None, logstdout=None): file = open(logfile, 'a') if logfile else sys.stderr log.startLogging(file, setStdout=logstdout) +def _switch_descriptors(): + global _prev_descriptors + + cur = (sys.stdout, sys.stderr) + sys.stdout, sys.stderr = _prev_descriptors + _prev_descriptors = cur + def msg(message, level=INFO, component=BOT_NAME, domain=None): """Log message according to the level""" dispatcher.send(signal=logmessage_received, message=message, level=level, \ diff --git a/scrapy/shell.py b/scrapy/shell.py index 0c579a40b..a137557cb 100644 --- a/scrapy/shell.py +++ b/scrapy/shell.py @@ -20,6 +20,7 @@ from scrapy.core.manager import scrapymanager from scrapy.core.engine import scrapyengine from scrapy.http import Request from scrapy.fetcher import get_or_create_spider +from scrapy import log def relevant_var(varname): return varname not in ['shelp', 'fetch', 'view', '__builtins__', 'In', \ @@ -40,10 +41,11 @@ class Shell(object): requires_project = False - def __init__(self, update_vars): + def __init__(self, update_vars=None, nofetch=False): self.vars = {} self.update_vars = update_vars self.item_class = load_object(settings['DEFAULT_ITEM_CLASS']) + self.nofetch = nofetch def fetch(self, request_or_url, print_help=False): if isinstance(request_or_url, Request): @@ -74,11 +76,13 @@ class Shell(object): self.vars['request'] = request self.vars['spider'] = spiders.fromurl(url) - self.vars['fetch'] = self.fetch + if not self.nofetch: + self.vars['fetch'] = self.fetch self.vars['view'] = open_in_browser self.vars['shelp'] = self.print_help - self.update_vars(self.vars) + if self.update_vars: + self.update_vars(self.vars) def print_help(self): print "Available objects" @@ -92,7 +96,8 @@ class Shell(object): print "===================" print print " shelp() : Prints this help." - print " fetch(req_or_url) : Fetch a new request or URL and update objects" + if not self.nofetch: + print " fetch(req_or_url) : Fetch a new request or URL and update objects" print " view(response) : View response in a browser" print @@ -103,12 +108,20 @@ class Shell(object): reactor.callInThread(self._console_thread, url) scrapymanager.start() - def _console_thread(self, url=None): - self.populate_vars() - if url: - result = self.fetch(url, print_help=True) - else: - self.print_help() + def inspect_response(self, response): + print + print "Scrapy Shell" + print "============" + print + print "Inspecting: %s" % response + print "Use shelp() to see available objects" + print + request = response.request + url = request.url + self.populate_vars(url, response, request) + self._run_console() + + def _run_console(self): try: # use IPython if available import IPython shell = IPython.Shell.IPShell(argv=[], user_ns=self.vars) @@ -124,4 +137,19 @@ class Shell(object): import rlcompleter readline.parse_and_bind("tab:complete") code.interact(local=self.vars) + + def _console_thread(self, url=None): + self.populate_vars() + if url: + result = self.fetch(url, print_help=True) + else: + self.print_help() + self._run_console() reactor.callFromThread(scrapymanager.stop) + +def inspect_response(response): + """Open a shell to inspect the given response""" + shell = Shell(nofetch=True) + log._switch_descriptors() + shell.inspect_response(response) + log._switch_descriptors()