mirror of https://github.com/scrapy/scrapy.git
Merge remote-tracking branch 'origin/master' into asyncio-base
This commit is contained in:
commit
20289be810
|
|
@ -0,0 +1,9 @@
|
|||
version: 2
|
||||
sphinx:
|
||||
configuration: docs/conf.py
|
||||
python:
|
||||
# For available versions, see:
|
||||
# https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image
|
||||
version: 3.7 # Keep in sync with .travis.yml
|
||||
install:
|
||||
- requirements: docs/requirements.txt
|
||||
|
|
@ -12,10 +12,9 @@ matrix:
|
|||
- env: TOXENV=flake8
|
||||
python: 3.8
|
||||
- env: TOXENV=pypy3
|
||||
python: 3.5
|
||||
- env: TOXENV=py35
|
||||
python: 3.5
|
||||
- env: TOXENV=py35-pinned
|
||||
- env: TOXENV=pinned
|
||||
python: 3.5
|
||||
- env: TOXENV=py35-asyncio
|
||||
python: 3.5
|
||||
|
|
@ -25,12 +24,12 @@ matrix:
|
|||
python: 3.7
|
||||
- env: TOXENV=py38
|
||||
python: 3.8
|
||||
- env: TOXENV=py38-extra-deps
|
||||
- env: TOXENV=extra-deps
|
||||
python: 3.8
|
||||
- env: TOXENV=py38-asyncio
|
||||
python: 3.8
|
||||
- env: TOXENV=docs
|
||||
python: 3.6
|
||||
python: 3.7 # Keep in sync with .readthedocs.yml
|
||||
install:
|
||||
- |
|
||||
if [ "$TOXENV" = "pypy3" ]; then
|
||||
|
|
|
|||
|
|
@ -0,0 +1,281 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>Quotes to Scrape</title>
|
||||
<link rel="stylesheet" href="/static/bootstrap.min.css">
|
||||
<link rel="stylesheet" href="/static/main.css">
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<div class="row header-box">
|
||||
<div class="col-md-8">
|
||||
<h1>
|
||||
<a href="/" style="text-decoration: none">Quotes to Scrape</a>
|
||||
</h1>
|
||||
</div>
|
||||
<div class="col-md-4">
|
||||
<p>
|
||||
|
||||
<a href="/login">Login</a>
|
||||
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
|
||||
<div class="row">
|
||||
<div class="col-md-8">
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
|
||||
|
||||
<a class="tag" href="/tag/change/page/1/">change</a>
|
||||
|
||||
<a class="tag" href="/tag/deep-thoughts/page/1/">deep-thoughts</a>
|
||||
|
||||
<a class="tag" href="/tag/thinking/page/1/">thinking</a>
|
||||
|
||||
<a class="tag" href="/tag/world/page/1/">world</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“It is our choices, Harry, that show what we truly are, far more than our abilities.”</span>
|
||||
<span>by <small class="author" itemprop="author">J.K. Rowling</small>
|
||||
<a href="/author/J-K-Rowling">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
|
||||
|
||||
<a class="tag" href="/tag/abilities/page/1/">abilities</a>
|
||||
|
||||
<a class="tag" href="/tag/choices/page/1/">choices</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
<a class="tag" href="/tag/life/page/1/">life</a>
|
||||
|
||||
<a class="tag" href="/tag/live/page/1/">live</a>
|
||||
|
||||
<a class="tag" href="/tag/miracle/page/1/">miracle</a>
|
||||
|
||||
<a class="tag" href="/tag/miracles/page/1/">miracles</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.”</span>
|
||||
<span>by <small class="author" itemprop="author">Jane Austen</small>
|
||||
<a href="/author/Jane-Austen">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
|
||||
|
||||
<a class="tag" href="/tag/aliteracy/page/1/">aliteracy</a>
|
||||
|
||||
<a class="tag" href="/tag/books/page/1/">books</a>
|
||||
|
||||
<a class="tag" href="/tag/classic/page/1/">classic</a>
|
||||
|
||||
<a class="tag" href="/tag/humor/page/1/">humor</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.”</span>
|
||||
<span>by <small class="author" itemprop="author">Marilyn Monroe</small>
|
||||
<a href="/author/Marilyn-Monroe">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
|
||||
|
||||
<a class="tag" href="/tag/be-yourself/page/1/">be-yourself</a>
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“Try not to become a man of success. Rather become a man of value.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
|
||||
|
||||
<a class="tag" href="/tag/adulthood/page/1/">adulthood</a>
|
||||
|
||||
<a class="tag" href="/tag/success/page/1/">success</a>
|
||||
|
||||
<a class="tag" href="/tag/value/page/1/">value</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“It is better to be hated for what you are than to be loved for what you are not.”</span>
|
||||
<span>by <small class="author" itemprop="author">André Gide</small>
|
||||
<a href="/author/Andre-Gide">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="life,love" / >
|
||||
|
||||
<a class="tag" href="/tag/life/page/1/">life</a>
|
||||
|
||||
<a class="tag" href="/tag/love/page/1/">love</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“I have not failed. I've just found 10,000 ways that won't work.”</span>
|
||||
<span>by <small class="author" itemprop="author">Thomas A. Edison</small>
|
||||
<a href="/author/Thomas-A-Edison">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
|
||||
|
||||
<a class="tag" href="/tag/edison/page/1/">edison</a>
|
||||
|
||||
<a class="tag" href="/tag/failure/page/1/">failure</a>
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
<a class="tag" href="/tag/paraphrased/page/1/">paraphrased</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“A woman is like a tea bag; you never know how strong it is until it's in hot water.”</span>
|
||||
<span>by <small class="author" itemprop="author">Eleanor Roosevelt</small>
|
||||
<a href="/author/Eleanor-Roosevelt">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
|
||||
|
||||
<a class="tag" href="/tag/misattributed-eleanor-roosevelt/page/1/">misattributed-eleanor-roosevelt</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“A day without sunshine is like, you know, night.”</span>
|
||||
<span>by <small class="author" itemprop="author">Steve Martin</small>
|
||||
<a href="/author/Steve-Martin">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
|
||||
|
||||
<a class="tag" href="/tag/humor/page/1/">humor</a>
|
||||
|
||||
<a class="tag" href="/tag/obvious/page/1/">obvious</a>
|
||||
|
||||
<a class="tag" href="/tag/simile/page/1/">simile</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<nav>
|
||||
<ul class="pager">
|
||||
|
||||
|
||||
<li class="next">
|
||||
<a href="/page/2/">Next <span aria-hidden="true">→</span></a>
|
||||
</li>
|
||||
|
||||
</ul>
|
||||
</nav>
|
||||
</div>
|
||||
<div class="col-md-4 tags-box">
|
||||
|
||||
<h2>Top Ten tags</h2>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 28px" href="/tag/love/">love</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 26px" href="/tag/inspirational/">inspirational</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 26px" href="/tag/life/">life</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 24px" href="/tag/humor/">humor</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 22px" href="/tag/books/">books</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 14px" href="/tag/reading/">reading</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 10px" href="/tag/friendship/">friendship</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 8px" href="/tag/friends/">friends</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 8px" href="/tag/truth/">truth</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 6px" href="/tag/simile/">simile</a>
|
||||
</span>
|
||||
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
<footer class="footer">
|
||||
<div class="container">
|
||||
<p class="text-muted">
|
||||
Quotes by: <a href="https://www.goodreads.com/quotes">GoodReads.com</a>
|
||||
</p>
|
||||
<p class="copyright">
|
||||
Made with <span class='sh-red'>❤</span> by <a href="https://scrapinghub.com">Scrapinghub</a>
|
||||
</p>
|
||||
</div>
|
||||
</footer>
|
||||
</body>
|
||||
</html>
|
||||
13
docs/conf.py
13
docs/conf.py
|
|
@ -12,6 +12,7 @@
|
|||
# serve to show the default.
|
||||
|
||||
import sys
|
||||
from datetime import datetime
|
||||
from os import path
|
||||
|
||||
# If your extensions are in another directory, add it here. If the directory
|
||||
|
|
@ -49,8 +50,8 @@ source_suffix = '.rst'
|
|||
master_doc = 'index'
|
||||
|
||||
# General information about the project.
|
||||
project = u'Scrapy'
|
||||
copyright = u'2008–2018, Scrapy developers'
|
||||
project = 'Scrapy'
|
||||
copyright = '2008–{}, Scrapy developers'.format(datetime.now().year)
|
||||
|
||||
# The version info for the project you're documenting, acts as replacement for
|
||||
# |version| and |release|, also used in various other places throughout the
|
||||
|
|
@ -194,8 +195,8 @@ htmlhelp_basename = 'Scrapydoc'
|
|||
# Grouping the document tree into LaTeX files. List of tuples
|
||||
# (source start file, target name, title, author, document class [howto/manual]).
|
||||
latex_documents = [
|
||||
('index', 'Scrapy.tex', u'Scrapy Documentation',
|
||||
u'Scrapy developers', 'manual'),
|
||||
('index', 'Scrapy.tex', 'Scrapy Documentation',
|
||||
'Scrapy developers', 'manual'),
|
||||
]
|
||||
|
||||
# The name of an image file (relative to this directory) to place at the top of
|
||||
|
|
@ -268,6 +269,10 @@ coverage_ignore_pyobjects = [
|
|||
|
||||
# Never documented before, and deprecated now.
|
||||
r'^scrapy\.item\.DictItem$',
|
||||
r'^scrapy\.linkextractors\.FilteringLinkExtractor$',
|
||||
|
||||
# Implementation detail of LxmlLinkExtractor
|
||||
r'^scrapy\.linkextractors\.lxmlhtml\.LxmlParserLinkExtractor',
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -44,7 +44,7 @@ guidelines when you're going to report a new bug.
|
|||
* check the :ref:`FAQ <faq>` first to see if your issue is addressed in a
|
||||
well-known question
|
||||
|
||||
* if you have a general question about scrapy usage, please ask it at
|
||||
* if you have a general question about Scrapy usage, please ask it at
|
||||
`Stack Overflow <https://stackoverflow.com/questions/tagged/scrapy>`__
|
||||
(use "scrapy" tag).
|
||||
|
||||
|
|
@ -203,17 +203,9 @@ Tests are implemented using the :doc:`Twisted unit-testing framework
|
|||
Running tests
|
||||
-------------
|
||||
|
||||
Make sure you have a recent enough :doc:`tox <tox:index>` installation:
|
||||
To run all tests::
|
||||
|
||||
``tox --version``
|
||||
|
||||
If your version is older than 1.7.0, please update it first:
|
||||
|
||||
``pip install -U tox``
|
||||
|
||||
To run all tests go to the root directory of Scrapy source code and run:
|
||||
|
||||
``tox``
|
||||
tox
|
||||
|
||||
To run a specific test (say ``tests/test_loader.py``) use:
|
||||
|
||||
|
|
@ -225,11 +217,11 @@ the tests with Python 3.6 use::
|
|||
|
||||
tox -e py36
|
||||
|
||||
You can also specify a comma-separated list of environmets, and use :ref:`tox’s
|
||||
You can also specify a comma-separated list of environments, and use :ref:`tox’s
|
||||
parallel mode <tox:parallel_mode>` to run the tests on multiple environments in
|
||||
parallel::
|
||||
|
||||
tox -e py27,py36 -p auto
|
||||
tox -e py36,py38 -p auto
|
||||
|
||||
To pass command-line options to :doc:`pytest <pytest:index>`, add them after
|
||||
``--`` in your call to :doc:`tox <tox:index>`. Using ``--`` overrides the
|
||||
|
|
|
|||
|
|
@ -338,7 +338,7 @@ How to split an item into multiple items in an item pipeline?
|
|||
input item. :ref:`Create a spider middleware <custom-spider-middleware>`
|
||||
instead, and use its
|
||||
:meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_spider_output`
|
||||
method for this puspose. For example::
|
||||
method for this purpose. For example::
|
||||
|
||||
from copy import deepcopy
|
||||
|
||||
|
|
|
|||
|
|
@ -170,7 +170,7 @@ Solving specific problems
|
|||
Get answers to most frequently asked questions.
|
||||
|
||||
:doc:`topics/debug`
|
||||
Learn how to debug common problems of your scrapy spider.
|
||||
Learn how to debug common problems of your Scrapy spider.
|
||||
|
||||
:doc:`topics/contracts`
|
||||
Learn how to use contracts for testing your spiders.
|
||||
|
|
|
|||
|
|
@ -78,9 +78,9 @@ TL;DR: We recommend installing Scrapy inside a virtual environment
|
|||
on all platforms.
|
||||
|
||||
Python packages can be installed either globally (a.k.a system wide),
|
||||
or in user-space. We do not recommend installing scrapy system wide.
|
||||
or in user-space. We do not recommend installing Scrapy system wide.
|
||||
|
||||
Instead, we recommend that you install scrapy within a so-called
|
||||
Instead, we recommend that you install Scrapy within a so-called
|
||||
"virtual environment" (`virtualenv`_).
|
||||
Virtualenvs allow you to not conflict with already-installed Python
|
||||
system packages (which could break some of your system tools and scripts),
|
||||
|
|
@ -97,7 +97,7 @@ Check this `user guide`_ on how to create your virtualenv.
|
|||
.. note::
|
||||
If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs.
|
||||
|
||||
Once you have created a virtualenv, you can install scrapy inside it with ``pip``,
|
||||
Once you have created a virtualenv, you can install Scrapy inside it with ``pip``,
|
||||
just like any other Python package.
|
||||
(See :ref:`platform-specific guides <intro-install-platform-notes>`
|
||||
below for non-Python dependencies that you may need to install beforehand).
|
||||
|
|
@ -144,7 +144,7 @@ albeit with potential issues with TLS connections.
|
|||
typically too old and slow to catch up with latest Scrapy.
|
||||
|
||||
|
||||
To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install
|
||||
To install Scrapy on Ubuntu (or Ubuntu-based) systems, you need to install
|
||||
these dependencies::
|
||||
|
||||
sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev
|
||||
|
|
@ -225,17 +225,17 @@ PyPy
|
|||
We recommend using the latest PyPy version. The version tested is 5.9.0.
|
||||
For PyPy3, only Linux installation was tested.
|
||||
|
||||
Most scrapy dependencides now have binary wheels for CPython, but not for PyPy.
|
||||
Most Scrapy dependencides now have binary wheels for CPython, but not for PyPy.
|
||||
This means that these dependecies will be built during installation.
|
||||
On OS X, you are likely to face an issue with building Cryptography dependency,
|
||||
solution to this problem is described
|
||||
`here <https://github.com/pyca/cryptography/issues/2692#issuecomment-272773481>`_,
|
||||
that is to ``brew install openssl`` and then export the flags that this command
|
||||
recommends (only needed when installing scrapy). Installing on Linux has no special
|
||||
recommends (only needed when installing Scrapy). Installing on Linux has no special
|
||||
issues besides installing build dependencies.
|
||||
Installing scrapy with PyPy on Windows is not tested.
|
||||
Installing Scrapy with PyPy on Windows is not tested.
|
||||
|
||||
You can check that scrapy is installed correctly by running ``scrapy bench``.
|
||||
You can check that Scrapy is installed correctly by running ``scrapy bench``.
|
||||
If this command gives errors such as
|
||||
``TypeError: ... got 2 unexpected keyword arguments``, this means
|
||||
that setuptools was unable to pick up one PyPy-specific dependency.
|
||||
|
|
|
|||
|
|
@ -252,30 +252,30 @@ The result of running ``response.css('title')`` is a list-like object called
|
|||
and allow you to run further queries to fine-grain the selection or extract the
|
||||
data.
|
||||
|
||||
To extract the text from the title above, you can do::
|
||||
To extract the text from the title above, you can do:
|
||||
|
||||
>>> response.css('title::text').getall()
|
||||
['Quotes to Scrape']
|
||||
>>> response.css('title::text').getall()
|
||||
['Quotes to Scrape']
|
||||
|
||||
There are two things to note here: one is that we've added ``::text`` to the
|
||||
CSS query, to mean we want to select only the text elements directly inside
|
||||
``<title>`` element. If we don't specify ``::text``, we'd get the full title
|
||||
element, including its tags::
|
||||
element, including its tags:
|
||||
|
||||
>>> response.css('title').getall()
|
||||
['<title>Quotes to Scrape</title>']
|
||||
>>> response.css('title').getall()
|
||||
['<title>Quotes to Scrape</title>']
|
||||
|
||||
The other thing is that the result of calling ``.getall()`` is a list: it is
|
||||
possible that a selector returns more than one result, so we extract them all.
|
||||
When you know you just want the first result, as in this case, you can do::
|
||||
When you know you just want the first result, as in this case, you can do:
|
||||
|
||||
>>> response.css('title::text').get()
|
||||
'Quotes to Scrape'
|
||||
>>> response.css('title::text').get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
As an alternative, you could've written::
|
||||
As an alternative, you could've written:
|
||||
|
||||
>>> response.css('title::text')[0].get()
|
||||
'Quotes to Scrape'
|
||||
>>> response.css('title::text')[0].get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
However, using ``.get()`` directly on a :class:`~scrapy.selector.SelectorList`
|
||||
instance avoids an ``IndexError`` and returns ``None`` when it doesn't
|
||||
|
|
@ -288,14 +288,14 @@ to be scraped, you can at least get **some** data.
|
|||
Besides the :meth:`~scrapy.selector.SelectorList.getall` and
|
||||
:meth:`~scrapy.selector.SelectorList.get` methods, you can also use
|
||||
the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular
|
||||
expressions`_::
|
||||
expressions`_:
|
||||
|
||||
>>> response.css('title::text').re(r'Quotes.*')
|
||||
['Quotes to Scrape']
|
||||
>>> response.css('title::text').re(r'Q\w+')
|
||||
['Quotes']
|
||||
>>> response.css('title::text').re(r'(\w+) to (\w+)')
|
||||
['Quotes', 'Scrape']
|
||||
>>> response.css('title::text').re(r'Quotes.*')
|
||||
['Quotes to Scrape']
|
||||
>>> response.css('title::text').re(r'Q\w+')
|
||||
['Quotes']
|
||||
>>> response.css('title::text').re(r'(\w+) to (\w+)')
|
||||
['Quotes', 'Scrape']
|
||||
|
||||
In order to find the proper CSS selectors to use, you might find useful opening
|
||||
the response page from the shell in your web browser using ``view(response)``.
|
||||
|
|
@ -312,12 +312,12 @@ visually selected elements, which works in many browsers.
|
|||
XPath: a brief intro
|
||||
^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions::
|
||||
Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions:
|
||||
|
||||
>>> response.xpath('//title')
|
||||
[<Selector xpath='//title' data='<title>Quotes to Scrape</title>'>]
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Quotes to Scrape'
|
||||
>>> response.xpath('//title')
|
||||
[<Selector xpath='//title' data='<title>Quotes to Scrape</title>'>]
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
XPath expressions are very powerful, and are the foundation of Scrapy
|
||||
Selectors. In fact, CSS selectors are converted to XPath under-the-hood. You
|
||||
|
|
@ -372,35 +372,35 @@ we want::
|
|||
|
||||
$ scrapy shell 'http://quotes.toscrape.com'
|
||||
|
||||
We get a list of selectors for the quote HTML elements with::
|
||||
We get a list of selectors for the quote HTML elements with:
|
||||
|
||||
>>> response.css("div.quote")
|
||||
[<Selector xpath="descendant-or-self::div[@class and contains(concat(' ', normalize-space(@class), ' '), ' quote ')]" data='<div class="quote" itemscope itemtype...'>,
|
||||
<Selector xpath="descendant-or-self::div[@class and contains(concat(' ', normalize-space(@class), ' '), ' quote ')]" data='<div class="quote" itemscope itemtype...'>,
|
||||
...]
|
||||
>>> response.css("div.quote")
|
||||
[<Selector xpath="descendant-or-self::div[@class and contains(concat(' ', normalize-space(@class), ' '), ' quote ')]" data='<div class="quote" itemscope itemtype...'>,
|
||||
<Selector xpath="descendant-or-self::div[@class and contains(concat(' ', normalize-space(@class), ' '), ' quote ')]" data='<div class="quote" itemscope itemtype...'>,
|
||||
...]
|
||||
|
||||
Each of the selectors returned by the query above allows us to run further
|
||||
queries over their sub-elements. Let's assign the first selector to a
|
||||
variable, so that we can run our CSS selectors directly on a particular quote::
|
||||
variable, so that we can run our CSS selectors directly on a particular quote:
|
||||
|
||||
>>> quote = response.css("div.quote")[0]
|
||||
>>> quote = response.css("div.quote")[0]
|
||||
|
||||
Now, let's extract ``text``, ``author`` and the ``tags`` from that quote
|
||||
using the ``quote`` object we just created::
|
||||
using the ``quote`` object we just created:
|
||||
|
||||
>>> text = quote.css("span.text::text").get()
|
||||
>>> text
|
||||
'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'
|
||||
>>> author = quote.css("small.author::text").get()
|
||||
>>> author
|
||||
'Albert Einstein'
|
||||
>>> text = quote.css("span.text::text").get()
|
||||
>>> text
|
||||
'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'
|
||||
>>> author = quote.css("small.author::text").get()
|
||||
>>> author
|
||||
'Albert Einstein'
|
||||
|
||||
Given that the tags are a list of strings, we can use the ``.getall()`` method
|
||||
to get all of them::
|
||||
to get all of them:
|
||||
|
||||
>>> tags = quote.css("div.tags a.tag::text").getall()
|
||||
>>> tags
|
||||
['change', 'deep-thoughts', 'thinking', 'world']
|
||||
>>> tags = quote.css("div.tags a.tag::text").getall()
|
||||
>>> tags
|
||||
['change', 'deep-thoughts', 'thinking', 'world']
|
||||
|
||||
.. invisible-code-block: python
|
||||
|
||||
|
|
@ -409,16 +409,16 @@ to get all of them::
|
|||
.. skip: next if(version_info < (3, 6), reason="Only Python 3.6+ dictionaries match the output")
|
||||
|
||||
Having figured out how to extract each bit, we can now iterate over all the
|
||||
quotes elements and put them together into a Python dictionary::
|
||||
quotes elements and put them together into a Python dictionary:
|
||||
|
||||
>>> for quote in response.css("div.quote"):
|
||||
... text = quote.css("span.text::text").get()
|
||||
... author = quote.css("small.author::text").get()
|
||||
... tags = quote.css("div.tags a.tag::text").getall()
|
||||
... print(dict(text=text, author=author, tags=tags))
|
||||
{'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']}
|
||||
{'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']}
|
||||
...
|
||||
>>> for quote in response.css("div.quote"):
|
||||
... text = quote.css("span.text::text").get()
|
||||
... author = quote.css("small.author::text").get()
|
||||
... tags = quote.css("div.tags a.tag::text").getall()
|
||||
... print(dict(text=text, author=author, tags=tags))
|
||||
{'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']}
|
||||
{'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']}
|
||||
...
|
||||
|
||||
Extracting data in our spider
|
||||
-----------------------------
|
||||
|
|
@ -516,23 +516,23 @@ markup:
|
|||
</li>
|
||||
</ul>
|
||||
|
||||
We can try extracting it in the shell::
|
||||
We can try extracting it in the shell:
|
||||
|
||||
>>> response.css('li.next a').get()
|
||||
'<a href="/page/2/">Next <span aria-hidden="true">→</span></a>'
|
||||
>>> response.css('li.next a').get()
|
||||
'<a href="/page/2/">Next <span aria-hidden="true">→</span></a>'
|
||||
|
||||
This gets the anchor element, but we want the attribute ``href``. For that,
|
||||
Scrapy supports a CSS extension that lets you select the attribute contents,
|
||||
like this::
|
||||
like this:
|
||||
|
||||
>>> response.css('li.next a::attr(href)').get()
|
||||
'/page/2/'
|
||||
>>> response.css('li.next a::attr(href)').get()
|
||||
'/page/2/'
|
||||
|
||||
There is also an ``attrib`` property available
|
||||
(see :ref:`selecting-attributes` for more)::
|
||||
(see :ref:`selecting-attributes` for more):
|
||||
|
||||
>>> response.css('li.next a').attrib['href']
|
||||
'/page/2/'
|
||||
>>> response.css('li.next a').attrib['href']
|
||||
'/page/2/'
|
||||
|
||||
Let's see now our spider modified to recursively follow the link to the next
|
||||
page, extracting data from it::
|
||||
|
|
|
|||
|
|
@ -47,13 +47,13 @@ Backward-incompatible changes
|
|||
(:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`)
|
||||
|
||||
* :class:`~scrapy.loader.ItemLoader` now turns the values of its input item
|
||||
into lists::
|
||||
into lists:
|
||||
|
||||
>>> item = MyItem()
|
||||
>>> item['field'] = 'value1'
|
||||
>>> loader = ItemLoader(item=item)
|
||||
>>> item['field']
|
||||
['value1']
|
||||
>>> item = MyItem()
|
||||
>>> item['field'] = 'value1'
|
||||
>>> loader = ItemLoader(item=item)
|
||||
>>> item['field']
|
||||
['value1']
|
||||
|
||||
This is needed to allow adding values to existing fields
|
||||
(``loader.add_value('field', 'value2')``).
|
||||
|
|
@ -678,7 +678,7 @@ Usability improvements
|
|||
* a message is added to IgnoreRequest in RobotsTxtMiddleware (:issue:`3113`)
|
||||
* better validation of ``url`` argument in ``Response.follow`` (:issue:`3131`)
|
||||
* non-zero exit code is returned from Scrapy commands when error happens
|
||||
on spider inititalization (:issue:`3226`)
|
||||
on spider initialization (:issue:`3226`)
|
||||
* Link extraction improvements: "ftp" is added to scheme list (:issue:`3152`);
|
||||
"flv" is added to common video extensions (:issue:`3165`)
|
||||
* better error message when an exporter is disabled (:issue:`3358`);
|
||||
|
|
@ -1156,7 +1156,7 @@ Bug fixes
|
|||
- Fix :command:`view` command ; it was a regression in v1.3.0 (:issue:`2503`).
|
||||
- Fix tests regarding ``*_EXPIRES settings`` with Files/Images pipelines (:issue:`2460`).
|
||||
- Fix name of generated pipeline class when using basic project template (:issue:`2466`).
|
||||
- Fix compatiblity with Twisted 17+ (:issue:`2496`, :issue:`2528`).
|
||||
- Fix compatibility with Twisted 17+ (:issue:`2496`, :issue:`2528`).
|
||||
- Fix ``scrapy.Item`` inheritance on Python 3.6 (:issue:`2511`).
|
||||
- Enforce numeric values for components order in ``SPIDER_MIDDLEWARES``,
|
||||
``DOWNLOADER_MIDDLEWARES``, ``EXTENIONS`` and ``SPIDER_CONTRACTS`` (:issue:`2420`).
|
||||
|
|
@ -1164,7 +1164,7 @@ Bug fixes
|
|||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
- Reword Code of Coduct section and upgrade to Contributor Covenant v1.4
|
||||
- Reword Code of Conduct section and upgrade to Contributor Covenant v1.4
|
||||
(:issue:`2469`).
|
||||
- Clarify that passing spider arguments converts them to spider attributes
|
||||
(:issue:`2483`).
|
||||
|
|
@ -1178,7 +1178,7 @@ Documentation
|
|||
Cleanups
|
||||
~~~~~~~~
|
||||
|
||||
- Remove reduntant check in ``MetaRefreshMiddleware`` (:issue:`2542`).
|
||||
- Remove redundant check in ``MetaRefreshMiddleware`` (:issue:`2542`).
|
||||
- Faster checks in ``LinkExtractor`` for allow/deny patterns (:issue:`2538`).
|
||||
- Remove dead code supporting old Twisted versions (:issue:`2544`).
|
||||
|
||||
|
|
@ -1204,7 +1204,7 @@ New Features
|
|||
- ``MailSender`` now accepts single strings as values for ``to`` and ``cc``
|
||||
arguments (:issue:`2272`)
|
||||
- ``scrapy fetch url``, ``scrapy shell url`` and ``fetch(url)`` inside
|
||||
scrapy shell now follow HTTP redirections by default (:issue:`2290`);
|
||||
Scrapy shell now follow HTTP redirections by default (:issue:`2290`);
|
||||
See :command:`fetch` and :command:`shell` for details.
|
||||
- ``HttpErrorMiddleware`` now logs errors with ``INFO`` level instead of ``DEBUG``;
|
||||
this is technically **backward incompatible** so please check your log parsers.
|
||||
|
|
@ -1360,7 +1360,7 @@ Documentation
|
|||
|
||||
- Grammar fixes: :issue:`2128`, :issue:`1566`.
|
||||
- Download stats badge removed from README (:issue:`2160`).
|
||||
- New scrapy :ref:`architecture diagram <topics-architecture>` (:issue:`2165`).
|
||||
- New Scrapy :ref:`architecture diagram <topics-architecture>` (:issue:`2165`).
|
||||
- Updated ``Response`` parameters documentation (:issue:`2197`).
|
||||
- Reworded misleading :setting:`RANDOMIZE_DOWNLOAD_DELAY` description (:issue:`2190`).
|
||||
- Add StackOverflow as a support channel (:issue:`2257`).
|
||||
|
|
@ -1450,7 +1450,7 @@ Documentation
|
|||
- Use "url" variable in downloader middleware example (:issue:`2015`)
|
||||
- Grammar fixes (:issue:`2054`, :issue:`2120`)
|
||||
- New FAQ entry on using BeautifulSoup in spider callbacks (:issue:`2048`)
|
||||
- Add notes about scrapy not working on Windows with Python 3 (:issue:`2060`)
|
||||
- Add notes about Scrapy not working on Windows with Python 3 (:issue:`2060`)
|
||||
- Encourage complete titles in pull requests (:issue:`2026`)
|
||||
|
||||
Tests
|
||||
|
|
@ -1509,7 +1509,7 @@ This 1.1 release brings a lot of interesting features and bug fixes:
|
|||
You can use :setting:`FILES_STORE_S3_ACL` to change it.
|
||||
- We've reimplemented ``canonicalize_url()`` for more correct output,
|
||||
especially for URLs with non-ASCII characters (:issue:`1947`).
|
||||
This could change link extractors output compared to previous scrapy versions.
|
||||
This could change link extractors output compared to previous Scrapy versions.
|
||||
This may also invalidate some cache entries you could still have from pre-1.1 runs.
|
||||
**Warning: backward incompatible!**.
|
||||
|
||||
|
|
@ -1705,7 +1705,7 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
- fix ValueError: Invalid XPath: //div/[id="not-exists"]/text() on selectors.rst (:commit:`ca8d60f`)
|
||||
- Typos corrections (:commit:`7067117`)
|
||||
- fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`)
|
||||
- Add note to ubuntu install section about debian compatibility (:commit:`23fda69`)
|
||||
- Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`)
|
||||
- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`)
|
||||
- Reference Homebrew's homepage for installation instructions (:commit:`1925db1`)
|
||||
- Add oldest supported tox version to contributing docs (:commit:`5d10d6d`)
|
||||
|
|
@ -1722,7 +1722,7 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
- Merge pull request #1513 from mgedmin/patch-2 (:commit:`5d4daf8`)
|
||||
- Typo (:commit:`f8d0682`)
|
||||
- Fix list formatting (:commit:`5f83a93`)
|
||||
- fix scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`)
|
||||
- fix Scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`)
|
||||
- Merge pull request #1475 from rweindl/patch-1 (:commit:`2d688cd`)
|
||||
- Update tutorial.rst (:commit:`fbc1f25`)
|
||||
- Merge pull request #1449 from rhoekman/patch-1 (:commit:`7d6538c`)
|
||||
|
|
@ -1734,7 +1734,7 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
Scrapy 1.0.3 (2015-08-11)
|
||||
-------------------------
|
||||
|
||||
- add service_identity to scrapy install_requires (:commit:`cbc2501`)
|
||||
- add service_identity to Scrapy install_requires (:commit:`cbc2501`)
|
||||
- Workaround for travis#296 (:commit:`66af9cd`)
|
||||
|
||||
.. _release-1.0.2:
|
||||
|
|
@ -1758,7 +1758,7 @@ Scrapy 1.0.1 (2015-07-01)
|
|||
- include tests/ to source distribution in MANIFEST.in (:commit:`eca227e`)
|
||||
- DOC Fix SelectJmes documentation (:commit:`b8567bc`)
|
||||
- DOC Bring Ubuntu and Archlinux outside of Windows subsection (:commit:`392233f`)
|
||||
- DOC remove version suffix from ubuntu package (:commit:`5303c66`)
|
||||
- DOC remove version suffix from Ubuntu package (:commit:`5303c66`)
|
||||
- DOC Update release date for 1.0 (:commit:`c89fa29`)
|
||||
|
||||
.. _release-1.0.0:
|
||||
|
|
@ -2211,7 +2211,7 @@ Scrapy 0.24.2 (2014-07-08)
|
|||
|
||||
- Use a mutable mapping to proxy deprecated settings.overrides and settings.defaults attribute (:commit:`e5e8133`)
|
||||
- there is not support for python3 yet (:commit:`3cd6146`)
|
||||
- Update python compatible version set to debian packages (:commit:`fa5d76b`)
|
||||
- Update python compatible version set to Debian packages (:commit:`fa5d76b`)
|
||||
- DOC fix formatting in release notes (:commit:`c6a9e20`)
|
||||
|
||||
Scrapy 0.24.1 (2014-06-27)
|
||||
|
|
@ -2229,12 +2229,12 @@ Enhancements
|
|||
|
||||
- Improve Scrapy top-level namespace (:issue:`494`, :issue:`684`)
|
||||
- Add selector shortcuts to responses (:issue:`554`, :issue:`690`)
|
||||
- Add new lxml based LinkExtractor to replace unmantained SgmlLinkExtractor
|
||||
- Add new lxml based LinkExtractor to replace unmaintained SgmlLinkExtractor
|
||||
(:issue:`559`, :issue:`761`, :issue:`763`)
|
||||
- Cleanup settings API - part of per-spider settings **GSoC project** (:issue:`737`)
|
||||
- Add UTF8 encoding header to templates (:issue:`688`, :issue:`762`)
|
||||
- Telnet console now binds to 127.0.0.1 by default (:issue:`699`)
|
||||
- Update debian/ubuntu install instructions (:issue:`509`, :issue:`549`)
|
||||
- Update Debian/Ubuntu install instructions (:issue:`509`, :issue:`549`)
|
||||
- Disable smart strings in lxml XPath evaluations (:issue:`535`)
|
||||
- Restore filesystem based cache as default for http
|
||||
cache middleware (:issue:`541`, :issue:`500`, :issue:`571`)
|
||||
|
|
@ -2267,7 +2267,7 @@ Enhancements
|
|||
- Tests and docs for ``request_fingerprint`` function (:issue:`597`)
|
||||
- Update SEP-19 for GSoC project ``per-spider settings`` (:issue:`705`)
|
||||
- Set exit code to non-zero when contracts fails (:issue:`727`)
|
||||
- Add a setting to control what class is instanciated as Downloader component
|
||||
- Add a setting to control what class is instantiated as Downloader component
|
||||
(:issue:`738`)
|
||||
- Pass response in ``item_dropped`` signal (:issue:`724`)
|
||||
- Improve ``scrapy check`` contracts command (:issue:`733`, :issue:`752`)
|
||||
|
|
@ -2276,7 +2276,7 @@ Enhancements
|
|||
- Add a note about reporting security issues (:issue:`697`)
|
||||
- Add LevelDB http cache storage backend (:issue:`626`, :issue:`500`)
|
||||
- Sort spider list output of ``scrapy list`` command (:issue:`742`)
|
||||
- Multiple documentation enhancemens and fixes
|
||||
- Multiple documentation enhancements and fixes
|
||||
(:issue:`575`, :issue:`587`, :issue:`590`, :issue:`596`, :issue:`610`,
|
||||
:issue:`617`, :issue:`618`, :issue:`627`, :issue:`613`, :issue:`643`,
|
||||
:issue:`654`, :issue:`675`, :issue:`663`, :issue:`711`, :issue:`714`)
|
||||
|
|
@ -2321,19 +2321,19 @@ Scrapy 0.22.1 (released 2014-02-08)
|
|||
- BaseSgmlLinkExtractor: Added unit test of a link with an inner tag (:commit:`c1cb418`)
|
||||
- BaseSgmlLinkExtractor: Fixed unknown_endtag() so that it only set current_link=None when the end tag match the opening tag (:commit:`7e4d627`)
|
||||
- Fix tests for Travis-CI build (:commit:`76c7e20`)
|
||||
- replace unencodeable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`)
|
||||
- replace unencodable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`)
|
||||
- RegexLinkExtractor: encode URL unicode value when creating Links (:commit:`d0ee545`)
|
||||
- Updated the tutorial crawl output with latest output. (:commit:`8da65de`)
|
||||
- Updated shell docs with the crawler reference and fixed the actual shell output. (:commit:`875b9ab`)
|
||||
- PEP8 minor edits. (:commit:`f89efaf`)
|
||||
- Expose current crawler in the scrapy shell. (:commit:`5349cec`)
|
||||
- Expose current crawler in the Scrapy shell. (:commit:`5349cec`)
|
||||
- Unused re import and PEP8 minor edits. (:commit:`387f414`)
|
||||
- Ignore None's values when using the ItemLoader. (:commit:`0632546`)
|
||||
- DOC Fixed HTTPCACHE_STORAGE typo in the default value which is now Filesystem instead Dbm. (:commit:`cde9a8c`)
|
||||
- show ubuntu setup instructions as literal code (:commit:`fb5c9c5`)
|
||||
- show Ubuntu setup instructions as literal code (:commit:`fb5c9c5`)
|
||||
- Update Ubuntu installation instructions (:commit:`70fb105`)
|
||||
- Merge pull request #550 from stray-leone/patch-1 (:commit:`6f70b6a`)
|
||||
- modify the version of scrapy ubuntu package (:commit:`725900d`)
|
||||
- modify the version of Scrapy Ubuntu package (:commit:`725900d`)
|
||||
- fix 0.22.0 release date (:commit:`af0219a`)
|
||||
- fix typos in news.rst and remove (not released yet) header (:commit:`b7f58f4`)
|
||||
|
||||
|
|
@ -2354,7 +2354,7 @@ Enhancements
|
|||
- Improve test coverage and forthcoming Python 3 support (:issue:`525`)
|
||||
- Promote startup info on settings and middleware to INFO level (:issue:`520`)
|
||||
- Support partials in ``get_func_args`` util (:issue:`506`, issue:`504`)
|
||||
- Allow running indiviual tests via tox (:issue:`503`)
|
||||
- Allow running individual tests via tox (:issue:`503`)
|
||||
- Update extensions ignored by link extractors (:issue:`498`)
|
||||
- Add middleware methods to get files/images/thumbs paths (:issue:`490`)
|
||||
- Improve offsite middleware tests (:issue:`478`)
|
||||
|
|
@ -2411,7 +2411,7 @@ Enhancements
|
|||
- scrapy.mail.MailSender now can connect over TLS or upgrade using STARTTLS (:issue:`327`)
|
||||
- New FilesPipeline with functionality factored out from ImagesPipeline (:issue:`370`, :issue:`409`)
|
||||
- Recommend Pillow instead of PIL for image handling (:issue:`317`)
|
||||
- Added debian packages for Ubuntu quantal and raring (:commit:`86230c0`)
|
||||
- Added Debian packages for Ubuntu Quantal and Raring (:commit:`86230c0`)
|
||||
- Mock server (used for tests) can listen for HTTPS requests (:issue:`410`)
|
||||
- Remove multi spider support from multiple core components
|
||||
(:issue:`422`, :issue:`421`, :issue:`420`, :issue:`419`, :issue:`423`, :issue:`418`)
|
||||
|
|
@ -2430,7 +2430,7 @@ Bugfixes
|
|||
- Fix tests under Django 1.6 (:commit:`b6bed44c`)
|
||||
- Lot of bugfixes to retry middleware under disconnections using HTTP 1.1 download handler
|
||||
- Fix inconsistencies among Twisted releases (:issue:`406`)
|
||||
- Fix scrapy shell bugs (:issue:`418`, :issue:`407`)
|
||||
- Fix Scrapy shell bugs (:issue:`418`, :issue:`407`)
|
||||
- Fix invalid variable name in setup.py (:issue:`429`)
|
||||
- Fix tutorial references (:issue:`387`)
|
||||
- Improve request-response docs (:issue:`391`)
|
||||
|
|
@ -2512,15 +2512,15 @@ Scrapy 0.18.1 (released 2013-08-27)
|
|||
- test PotentiaDataLoss errors on unbound responses (:commit:`b15470d`)
|
||||
- Treat responses without content-length or Transfer-Encoding as good responses (:commit:`c4bf324`)
|
||||
- do no include ResponseFailed if http11 handler is not enabled (:commit:`6cbe684`)
|
||||
- New HTTP client wraps connection losts in ResponseFailed exception. fix #373 (:commit:`1a20bba`)
|
||||
- New HTTP client wraps connection lost in ResponseFailed exception. fix #373 (:commit:`1a20bba`)
|
||||
- limit travis-ci build matrix (:commit:`3b01bb8`)
|
||||
- Merge pull request #375 from peterarenot/patch-1 (:commit:`fa766d7`)
|
||||
- Fixed so it refers to the correct folder (:commit:`3283809`)
|
||||
- added quantal & raring to support ubuntu releases (:commit:`1411923`)
|
||||
- added Quantal & Raring to support Ubuntu releases (:commit:`1411923`)
|
||||
- fix retry middleware which didn't retry certain connection errors after the upgrade to http1 client, closes GH-373 (:commit:`bb35ed0`)
|
||||
- fix XmlItemExporter in Python 2.7.4 and 2.7.5 (:commit:`de3e451`)
|
||||
- minor updates to 0.18 release notes (:commit:`c45e5f1`)
|
||||
- fix contributters list format (:commit:`0b60031`)
|
||||
- fix contributors list format (:commit:`0b60031`)
|
||||
|
||||
Scrapy 0.18.0 (released 2013-08-09)
|
||||
-----------------------------------
|
||||
|
|
@ -2555,8 +2555,8 @@ Scrapy 0.18.0 (released 2013-08-09)
|
|||
- Collect idle downloader slots (:issue:`297`)
|
||||
- Add ``ftp://`` scheme downloader handler (:issue:`329`)
|
||||
- Added downloader benchmark webserver and spider tools :ref:`benchmarking`
|
||||
- Moved persistent (on disk) queues to a separate project (queuelib_) which scrapy now depends on
|
||||
- Add scrapy commands using external libraries (:issue:`260`)
|
||||
- Moved persistent (on disk) queues to a separate project (queuelib_) which Scrapy now depends on
|
||||
- Add Scrapy commands using external libraries (:issue:`260`)
|
||||
- Added ``--pdb`` option to ``scrapy`` command line tool
|
||||
- Added :meth:`XPathSelector.remove_namespaces <scrapy.selector.Selector.remove_namespaces>` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`.
|
||||
- Several improvements to spider contracts
|
||||
|
|
@ -2568,7 +2568,7 @@ Scrapy 0.18.0 (released 2013-08-09)
|
|||
- several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez)
|
||||
- support custom download slots
|
||||
- added --spider option to "shell" command.
|
||||
- log overridden settings when scrapy starts
|
||||
- log overridden settings when Scrapy starts
|
||||
|
||||
Thanks to everyone who contribute to this release. Here is a list of
|
||||
contributors sorted by number of commits::
|
||||
|
|
@ -2617,7 +2617,7 @@ contributors sorted by number of commits::
|
|||
Scrapy 0.16.5 (released 2013-05-30)
|
||||
-----------------------------------
|
||||
|
||||
- obey request method when scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`)
|
||||
- obey request method when Scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`)
|
||||
- fix inaccurate downloader middleware documentation. refs #280 (:commit:`40667cb`)
|
||||
- doc: remove links to diveintopython.org, which is no longer available. closes #246 (:commit:`bd58bfa`)
|
||||
- Find form nodes in invalid html5 documents (:commit:`e3d6945`)
|
||||
|
|
@ -2631,8 +2631,8 @@ Scrapy 0.16.4 (released 2013-01-23)
|
|||
- Fixed error message formatting. log.err() doesn't support cool formatting and when error occurred, the message was: "ERROR: Error processing %(item)s" (:commit:`c16150c`)
|
||||
- lint and improve images pipeline error logging (:commit:`56b45fc`)
|
||||
- fixed doc typos (:commit:`243be84`)
|
||||
- add documentation topics: Broad Crawls & Common Practies (:commit:`1fbb715`)
|
||||
- fix bug in scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`)
|
||||
- add documentation topics: Broad Crawls & Common Practices (:commit:`1fbb715`)
|
||||
- fix bug in Scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`)
|
||||
- Update docs/topics/commands.rst (:commit:`28eac7a`)
|
||||
|
||||
Scrapy 0.16.3 (released 2012-12-07)
|
||||
|
|
@ -2651,11 +2651,11 @@ Scrapy 0.16.3 (released 2012-12-07)
|
|||
Scrapy 0.16.2 (released 2012-11-09)
|
||||
-----------------------------------
|
||||
|
||||
- scrapy contracts: python2.6 compat (:commit:`a4a9199`)
|
||||
- scrapy contracts verbose option (:commit:`ec41673`)
|
||||
- proper unittest-like output for scrapy contracts (:commit:`86635e4`)
|
||||
- Scrapy contracts: python2.6 compat (:commit:`a4a9199`)
|
||||
- Scrapy contracts verbose option (:commit:`ec41673`)
|
||||
- proper unittest-like output for Scrapy contracts (:commit:`86635e4`)
|
||||
- added open_in_browser to debugging doc (:commit:`c9b690d`)
|
||||
- removed reference to global scrapy stats from settings doc (:commit:`dd55067`)
|
||||
- removed reference to global Scrapy stats from settings doc (:commit:`dd55067`)
|
||||
- Fix SpiderState bug in Windows platforms (:commit:`58998f4`)
|
||||
|
||||
|
||||
|
|
@ -2665,7 +2665,7 @@ Scrapy 0.16.1 (released 2012-10-26)
|
|||
- fixed LogStats extension, which got broken after a wrong merge before the 0.16 release (:commit:`8c780fd`)
|
||||
- better backward compatibility for scrapy.conf.settings (:commit:`3403089`)
|
||||
- extended documentation on how to access crawler stats from extensions (:commit:`c4da0b5`)
|
||||
- removed .hgtags (no longer needed now that scrapy uses git) (:commit:`d52c188`)
|
||||
- removed .hgtags (no longer needed now that Scrapy uses git) (:commit:`d52c188`)
|
||||
- fix dashes under rst headers (:commit:`fa4f7f9`)
|
||||
- set release date for 0.16.0 in news (:commit:`e292246`)
|
||||
|
||||
|
|
@ -2680,8 +2680,7 @@ Scrapy changes:
|
|||
- documented :doc:`topics/autothrottle` and added to extensions installed by default. You still need to enable it with :setting:`AUTOTHROTTLE_ENABLED`
|
||||
- major Stats Collection refactoring: removed separation of global/per-spider stats, removed stats-related signals (``stats_spider_opened``, etc). Stats are much simpler now, backward compatibility is kept on the Stats Collector API and signals.
|
||||
- added :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_start_requests` method to spider middlewares
|
||||
- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info.
|
||||
- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info.
|
||||
- dropped Signals singleton. Signals should now be accessed through the Crawler.signals attribute. See the signals documentation for more info.
|
||||
- dropped Stats Collector singleton. Stats can now be accessed through the Crawler.stats attribute. See the stats collection documentation for more info.
|
||||
- documented :ref:`topics-api`
|
||||
- ``lxml`` is now the default selectors backend instead of ``libxml2``
|
||||
|
|
@ -2715,7 +2714,7 @@ Scrapy changes:
|
|||
Scrapy 0.14.4
|
||||
-------------
|
||||
|
||||
- added precise to supported ubuntu distros (:commit:`b7e46df`)
|
||||
- added precise to supported Ubuntu distros (:commit:`b7e46df`)
|
||||
- fixed bug in json-rpc webservice reported in https://groups.google.com/forum/#!topic/scrapy-users/qgVBmFybNAQ/discussion. also removed no longer supported 'run' command from extras/scrapy-ws.py (:commit:`340fbdb`)
|
||||
- meta tag attributes for content-type http equiv can be in any order. #123 (:commit:`0cb68af`)
|
||||
- replace "import Image" by more standard "from PIL import Image". closes #88 (:commit:`4d17048`)
|
||||
|
|
@ -2728,11 +2727,11 @@ Scrapy 0.14.3
|
|||
- include egg files used by testsuite in source distribution. #118 (:commit:`c897793`)
|
||||
- update docstring in project template to avoid confusion with genspider command, which may be considered as an advanced feature. refs #107 (:commit:`2548dcc`)
|
||||
- added note to docs/topics/firebug.rst about google directory being shut down (:commit:`668e352`)
|
||||
- dont discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`)
|
||||
- don't discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`)
|
||||
- do not fail handling unicode xpaths in libxml2 backed selectors (:commit:`b830e95`)
|
||||
- fixed minor mistake in Request objects documentation (:commit:`bf3c9ee`)
|
||||
- fixed minor defect in link extractors documentation (:commit:`ba14f38`)
|
||||
- removed some obsolete remaining code related to sqlite support in scrapy (:commit:`0665175`)
|
||||
- removed some obsolete remaining code related to sqlite support in Scrapy (:commit:`0665175`)
|
||||
|
||||
Scrapy 0.14.2
|
||||
-------------
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
-r ../requirements-py3.txt
|
||||
Sphinx>=2.1
|
||||
sphinx-hoverxref
|
||||
sphinx-notfound-page
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ Design goals
|
|||
============
|
||||
|
||||
1. be nicer to sites instead of using default download delay of zero
|
||||
2. automatically adjust scrapy to the optimum crawling speed, so the user
|
||||
2. automatically adjust Scrapy to the optimum crawling speed, so the user
|
||||
doesn't have to tune the download delays to find the optimum one.
|
||||
The user only needs to specify the maximum concurrent requests
|
||||
it allows, and the extension does the rest.
|
||||
|
|
|
|||
|
|
@ -29,7 +29,7 @@ in standard locations:
|
|||
1. ``/etc/scrapy.cfg`` or ``c:\scrapy\scrapy.cfg`` (system-wide),
|
||||
2. ``~/.config/scrapy.cfg`` (``$XDG_CONFIG_HOME``) and ``~/.scrapy.cfg`` (``$HOME``)
|
||||
for global (user-wide) settings, and
|
||||
3. ``scrapy.cfg`` inside a scrapy project's root (see next section).
|
||||
3. ``scrapy.cfg`` inside a Scrapy project's root (see next section).
|
||||
|
||||
Settings from these files are merged in the listed order of preference:
|
||||
user-defined values have higher priority than system-wide defaults
|
||||
|
|
|
|||
|
|
@ -64,7 +64,7 @@ Use the :command:`check` command to run the contract checks.
|
|||
Custom Contracts
|
||||
================
|
||||
|
||||
If you find you need more power than the built-in scrapy contracts you can
|
||||
If you find you need more power than the built-in Scrapy contracts you can
|
||||
create and load your own contracts in the project by using the
|
||||
:setting:`SPIDER_CONTRACTS` setting::
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ Debugging Spiders
|
|||
=================
|
||||
|
||||
This document explains the most common techniques for debugging spiders.
|
||||
Consider the following scrapy spider below::
|
||||
Consider the following Scrapy spider below::
|
||||
|
||||
import scrapy
|
||||
from myproject.items import MyItem
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ Therefore, you should keep in mind the following things:
|
|||
.. _topics-inspector:
|
||||
|
||||
Inspecting a website
|
||||
===================================
|
||||
====================
|
||||
|
||||
By far the most handy feature of the Developer Tools is the `Inspector`
|
||||
feature, which allows you to inspect the underlying HTML code of
|
||||
|
|
@ -79,13 +79,23 @@ sections and tags of a webpage, which greatly improves readability. You can
|
|||
expand and collapse a tag by clicking on the arrow in front of it or by double
|
||||
clicking directly on the tag. If we expand the ``span`` tag with the ``class=
|
||||
"text"`` we will see the quote-text we clicked on. The `Inspector` lets you
|
||||
copy XPaths to selected elements. Let's try it out: Right-click on the ``span``
|
||||
tag, select ``Copy > XPath`` and paste it in the scrapy shell like so::
|
||||
copy XPaths to selected elements. Let's try it out.
|
||||
|
||||
First open the Scrapy shell at http://quotes.toscrape.com/ in a terminal:
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
$ scrapy shell "http://quotes.toscrape.com/"
|
||||
(...)
|
||||
>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall()
|
||||
['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”]
|
||||
|
||||
Then, back to your web browser, right-click on the ``span`` tag, select
|
||||
``Copy > XPath`` and paste it in the Scrapy shell like so:
|
||||
|
||||
.. invisible-code-block: python
|
||||
|
||||
response = load_response('http://quotes.toscrape.com/', 'quotes.html')
|
||||
|
||||
>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall()
|
||||
['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”']
|
||||
|
||||
Adding ``text()`` at the end we are able to extract the first quote with this
|
||||
basic selector. But this XPath is not really that clever. All it does is
|
||||
|
|
@ -112,13 +122,13 @@ see each quote:
|
|||
|
||||
With this knowledge we can refine our XPath: Instead of a path to follow,
|
||||
we'll simply select all ``span`` tags with the ``class="text"`` by using
|
||||
the `has-class-extension`_::
|
||||
the `has-class-extension`_:
|
||||
|
||||
>>> response.xpath('//span[has-class("text")]/text()').getall()
|
||||
['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”,
|
||||
'“It is our choices, Harry, that show what we truly are, far more than our abilities.”',
|
||||
'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”',
|
||||
(...)]
|
||||
>>> response.xpath('//span[has-class("text")]/text()').getall()
|
||||
['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”',
|
||||
'“It is our choices, Harry, that show what we truly are, far more than our abilities.”',
|
||||
'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”',
|
||||
...]
|
||||
|
||||
And with one simple, cleverer XPath we are able to extract all quotes from
|
||||
the page. We could have constructed a loop over our first XPath to increase
|
||||
|
|
@ -159,7 +169,11 @@ The page is quite similar to the basic `quotes.toscrape.com`_-page,
|
|||
but instead of the above-mentioned ``Next`` button, the page
|
||||
automatically loads new quotes when you scroll to the bottom. We
|
||||
could go ahead and try out different XPaths directly, but instead
|
||||
we'll check another quite useful command from the scrapy shell::
|
||||
we'll check another quite useful command from the Scrapy shell:
|
||||
|
||||
.. skip: next
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
$ scrapy shell "quotes.toscrape.com/scroll"
|
||||
(...)
|
||||
|
|
|
|||
|
|
@ -172,27 +172,27 @@ data from it:
|
|||
data in JSON format, which you can then parse with `json.loads`_.
|
||||
|
||||
For example, if the JavaScript code contains a separate line like
|
||||
``var data = {"field": "value"};`` you can extract that data as follows::
|
||||
``var data = {"field": "value"};`` you can extract that data as follows:
|
||||
|
||||
>>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n'
|
||||
>>> json_data = response.css('script::text').re_first(pattern)
|
||||
>>> json.loads(json_data)
|
||||
{'field': 'value'}
|
||||
>>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n'
|
||||
>>> json_data = response.css('script::text').re_first(pattern)
|
||||
>>> json.loads(json_data)
|
||||
{'field': 'value'}
|
||||
|
||||
- Otherwise, use js2xml_ to convert the JavaScript code into an XML document
|
||||
that you can parse using :ref:`selectors <topics-selectors>`.
|
||||
|
||||
For example, if the JavaScript code contains
|
||||
``var data = {field: "value"};`` you can extract that data as follows::
|
||||
``var data = {field: "value"};`` you can extract that data as follows:
|
||||
|
||||
>>> import js2xml
|
||||
>>> import lxml.etree
|
||||
>>> from parsel import Selector
|
||||
>>> javascript = response.css('script::text').get()
|
||||
>>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode')
|
||||
>>> selector = Selector(text=xml)
|
||||
>>> selector.css('var[name="data"]').get()
|
||||
'<var name="data"><object><property name="field"><string>value</string></property></object></var>'
|
||||
>>> import js2xml
|
||||
>>> import lxml.etree
|
||||
>>> from parsel import Selector
|
||||
>>> javascript = response.css('script::text').get()
|
||||
>>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode')
|
||||
>>> selector = Selector(text=xml)
|
||||
>>> selector.css('var[name="data"]').get()
|
||||
'<var name="data"><object><property name="field"><string>value</string></property></object></var>'
|
||||
|
||||
.. _topics-javascript-rendering:
|
||||
|
||||
|
|
|
|||
|
|
@ -84,77 +84,74 @@ notice the API is very similar to the `dict API`_.
|
|||
Creating items
|
||||
--------------
|
||||
|
||||
::
|
||||
>>> product = Product(name='Desktop PC', price=1000)
|
||||
>>> print(product)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
>>> product = Product(name='Desktop PC', price=1000)
|
||||
>>> print(product)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
Getting field values
|
||||
--------------------
|
||||
|
||||
::
|
||||
>>> product['name']
|
||||
Desktop PC
|
||||
>>> product.get('name')
|
||||
Desktop PC
|
||||
|
||||
>>> product['name']
|
||||
Desktop PC
|
||||
>>> product.get('name')
|
||||
Desktop PC
|
||||
>>> product['price']
|
||||
1000
|
||||
|
||||
>>> product['price']
|
||||
1000
|
||||
>>> product['last_updated']
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'last_updated'
|
||||
|
||||
>>> product['last_updated']
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'last_updated'
|
||||
>>> product.get('last_updated', 'not set')
|
||||
not set
|
||||
|
||||
>>> product.get('last_updated', 'not set')
|
||||
not set
|
||||
>>> product['lala'] # getting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'lala'
|
||||
|
||||
>>> product['lala'] # getting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'lala'
|
||||
>>> product.get('lala', 'unknown field')
|
||||
'unknown field'
|
||||
|
||||
>>> product.get('lala', 'unknown field')
|
||||
'unknown field'
|
||||
>>> 'name' in product # is name field populated?
|
||||
True
|
||||
|
||||
>>> 'name' in product # is name field populated?
|
||||
True
|
||||
>>> 'last_updated' in product # is last_updated populated?
|
||||
False
|
||||
|
||||
>>> 'last_updated' in product # is last_updated populated?
|
||||
False
|
||||
>>> 'last_updated' in product.fields # is last_updated a declared field?
|
||||
True
|
||||
|
||||
>>> 'last_updated' in product.fields # is last_updated a declared field?
|
||||
True
|
||||
>>> 'lala' in product.fields # is lala a declared field?
|
||||
False
|
||||
|
||||
>>> 'lala' in product.fields # is lala a declared field?
|
||||
False
|
||||
|
||||
Setting field values
|
||||
--------------------
|
||||
|
||||
::
|
||||
>>> product['last_updated'] = 'today'
|
||||
>>> product['last_updated']
|
||||
today
|
||||
|
||||
>>> product['last_updated'] = 'today'
|
||||
>>> product['last_updated']
|
||||
today
|
||||
>>> product['lala'] = 'test' # setting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
>>> product['lala'] = 'test' # setting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
Accessing all populated values
|
||||
------------------------------
|
||||
|
||||
To access all populated values, just use the typical `dict API`_::
|
||||
To access all populated values, just use the typical `dict API`_:
|
||||
|
||||
>>> product.keys()
|
||||
['price', 'name']
|
||||
>>> product.keys()
|
||||
['price', 'name']
|
||||
|
||||
>>> product.items()
|
||||
[('price', 1000), ('name', 'Desktop PC')]
|
||||
>>> product.items()
|
||||
[('price', 1000), ('name', 'Desktop PC')]
|
||||
|
||||
|
||||
.. _copying-items:
|
||||
|
|
@ -194,20 +191,21 @@ To create a deep copy, call :meth:`~scrapy.item.Item.deepcopy` instead
|
|||
Other common tasks
|
||||
------------------
|
||||
|
||||
Creating dicts from items::
|
||||
Creating dicts from items:
|
||||
|
||||
>>> dict(product) # create a dict from all populated values
|
||||
{'price': 1000, 'name': 'Desktop PC'}
|
||||
>>> dict(product) # create a dict from all populated values
|
||||
{'price': 1000, 'name': 'Desktop PC'}
|
||||
|
||||
Creating items from dicts::
|
||||
Creating items from dicts:
|
||||
|
||||
>>> Product({'name': 'Laptop PC', 'price': 1500})
|
||||
Product(price=1500, name='Laptop PC')
|
||||
>>> Product({'name': 'Laptop PC', 'price': 1500})
|
||||
Product(price=1500, name='Laptop PC')
|
||||
|
||||
>>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
>>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
Extending Items
|
||||
===============
|
||||
|
|
|
|||
|
|
@ -74,7 +74,7 @@ Request serialization
|
|||
For persistence to work, :class:`~scrapy.http.Request` objects must be
|
||||
serializable with :mod:`pickle`, except for the ``callback`` and ``errback``
|
||||
values passed to their ``__init__`` method, which must be methods of the
|
||||
runnning :class:`~scrapy.spiders.Spider` class.
|
||||
running :class:`~scrapy.spiders.Spider` class.
|
||||
|
||||
If you wish to log the requests that couldn't be serialized, you can set the
|
||||
:setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page.
|
||||
|
|
|
|||
|
|
@ -110,7 +110,7 @@ ties the response lifetime to the requests' one, and that would definitely
|
|||
cause memory leaks.
|
||||
|
||||
Let's see how we can discover the cause (without knowing it
|
||||
a-priori, of course) by using the ``trackref`` tool.
|
||||
a priori, of course) by using the ``trackref`` tool.
|
||||
|
||||
After the crawler is running for a few minutes and we notice its memory usage
|
||||
has grown a lot, we can enter its telnet console and check the live
|
||||
|
|
@ -132,21 +132,21 @@ and check the code of the spider to discover the nasty line that is
|
|||
generating the leaks (passing response references inside requests).
|
||||
|
||||
Sometimes extra information about live objects can be helpful.
|
||||
Let's check the oldest response::
|
||||
Let's check the oldest response:
|
||||
|
||||
>>> from scrapy.utils.trackref import get_oldest
|
||||
>>> r = get_oldest('HtmlResponse')
|
||||
>>> r.url
|
||||
'http://www.somenastyspider.com/product.php?pid=123'
|
||||
>>> from scrapy.utils.trackref import get_oldest
|
||||
>>> r = get_oldest('HtmlResponse')
|
||||
>>> r.url
|
||||
'http://www.somenastyspider.com/product.php?pid=123'
|
||||
|
||||
If you want to iterate over all objects, instead of getting the oldest one, you
|
||||
can use the :func:`scrapy.utils.trackref.iter_all` function::
|
||||
can use the :func:`scrapy.utils.trackref.iter_all` function:
|
||||
|
||||
>>> from scrapy.utils.trackref import iter_all
|
||||
>>> [r.url for r in iter_all('HtmlResponse')]
|
||||
['http://www.somenastyspider.com/product.php?pid=123',
|
||||
'http://www.somenastyspider.com/product.php?pid=584',
|
||||
...
|
||||
>>> from scrapy.utils.trackref import iter_all
|
||||
>>> [r.url for r in iter_all('HtmlResponse')]
|
||||
['http://www.somenastyspider.com/product.php?pid=123',
|
||||
'http://www.somenastyspider.com/product.php?pid=584',
|
||||
...]
|
||||
|
||||
Too many spiders?
|
||||
-----------------
|
||||
|
|
@ -155,10 +155,10 @@ If your project has too many spiders executed in parallel,
|
|||
the output of :func:`prefs()` can be difficult to read.
|
||||
For this reason, that function has a ``ignore`` argument which can be used to
|
||||
ignore a particular class (and all its subclases). For
|
||||
example, this won't show any live references to spiders::
|
||||
example, this won't show any live references to spiders:
|
||||
|
||||
>>> from scrapy.spiders import Spider
|
||||
>>> prefs(ignore=Spider)
|
||||
>>> from scrapy.spiders import Spider
|
||||
>>> prefs(ignore=Spider)
|
||||
|
||||
.. module:: scrapy.utils.trackref
|
||||
:synopsis: Track references of live objects
|
||||
|
|
@ -214,41 +214,41 @@ If you use ``pip``, you can install Guppy with the following command::
|
|||
|
||||
The telnet console also comes with a built-in shortcut (``hpy``) for accessing
|
||||
Guppy heap objects. Here's an example to view all Python objects available in
|
||||
the heap using Guppy::
|
||||
the heap using Guppy:
|
||||
|
||||
>>> x = hpy.heap()
|
||||
>>> x.bytype
|
||||
Partition of a set of 297033 objects. Total size = 52587824 bytes.
|
||||
Index Count % Size % Cumulative % Type
|
||||
0 22307 8 16423880 31 16423880 31 dict
|
||||
1 122285 41 12441544 24 28865424 55 str
|
||||
2 68346 23 5966696 11 34832120 66 tuple
|
||||
3 227 0 5836528 11 40668648 77 unicode
|
||||
4 2461 1 2222272 4 42890920 82 type
|
||||
5 16870 6 2024400 4 44915320 85 function
|
||||
6 13949 5 1673880 3 46589200 89 types.CodeType
|
||||
7 13422 5 1653104 3 48242304 92 list
|
||||
8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern
|
||||
9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers
|
||||
<1676 more rows. Type e.g. '_.more' to view.>
|
||||
>>> x = hpy.heap()
|
||||
>>> x.bytype
|
||||
Partition of a set of 297033 objects. Total size = 52587824 bytes.
|
||||
Index Count % Size % Cumulative % Type
|
||||
0 22307 8 16423880 31 16423880 31 dict
|
||||
1 122285 41 12441544 24 28865424 55 str
|
||||
2 68346 23 5966696 11 34832120 66 tuple
|
||||
3 227 0 5836528 11 40668648 77 unicode
|
||||
4 2461 1 2222272 4 42890920 82 type
|
||||
5 16870 6 2024400 4 44915320 85 function
|
||||
6 13949 5 1673880 3 46589200 89 types.CodeType
|
||||
7 13422 5 1653104 3 48242304 92 list
|
||||
8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern
|
||||
9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers
|
||||
<1676 more rows. Type e.g. '_.more' to view.>
|
||||
|
||||
You can see that most space is used by dicts. Then, if you want to see from
|
||||
which attribute those dicts are referenced, you could do::
|
||||
which attribute those dicts are referenced, you could do:
|
||||
|
||||
>>> x.bytype[0].byvia
|
||||
Partition of a set of 22307 objects. Total size = 16423880 bytes.
|
||||
Index Count % Size % Cumulative % Referred Via:
|
||||
0 10982 49 9416336 57 9416336 57 '.__dict__'
|
||||
1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals'
|
||||
2 3097 14 1122904 7 13220744 80
|
||||
3 990 4 277200 2 13497944 82 "['cookies']"
|
||||
4 987 4 276360 2 13774304 84 "['cache']"
|
||||
5 985 4 275800 2 14050104 86 "['meta']"
|
||||
6 897 4 251160 2 14301264 87 '[2]'
|
||||
7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']"
|
||||
8 672 3 188160 1 14686312 89 "['cb_kwargs']"
|
||||
9 27 0 155016 1 14841328 90 '[1]'
|
||||
<333 more rows. Type e.g. '_.more' to view.>
|
||||
>>> x.bytype[0].byvia
|
||||
Partition of a set of 22307 objects. Total size = 16423880 bytes.
|
||||
Index Count % Size % Cumulative % Referred Via:
|
||||
0 10982 49 9416336 57 9416336 57 '.__dict__'
|
||||
1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals'
|
||||
2 3097 14 1122904 7 13220744 80
|
||||
3 990 4 277200 2 13497944 82 "['cookies']"
|
||||
4 987 4 276360 2 13774304 84 "['cache']"
|
||||
5 985 4 275800 2 14050104 86 "['meta']"
|
||||
6 897 4 251160 2 14301264 87 '[2]'
|
||||
7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']"
|
||||
8 672 3 188160 1 14686312 89 "['cb_kwargs']"
|
||||
9 27 0 155016 1 14841328 90 '[1]'
|
||||
<333 more rows. Type e.g. '_.more' to view.>
|
||||
|
||||
As you can see, the Guppy module is very powerful but also requires some deep
|
||||
knowledge about Python internals. For more info about Guppy, refer to the
|
||||
|
|
@ -269,32 +269,32 @@ If you use ``pip``, you can install muppy with the following command::
|
|||
pip install Pympler
|
||||
|
||||
Here's an example to view all Python objects available in
|
||||
the heap using muppy::
|
||||
the heap using muppy:
|
||||
|
||||
>>> from pympler import muppy
|
||||
>>> all_objects = muppy.get_objects()
|
||||
>>> len(all_objects)
|
||||
28667
|
||||
>>> from pympler import summary
|
||||
>>> suml = summary.summarize(all_objects)
|
||||
>>> summary.print_(suml)
|
||||
types | # objects | total size
|
||||
==================================== | =========== | ============
|
||||
<class 'str | 9822 | 1.10 MB
|
||||
<class 'dict | 1658 | 856.62 KB
|
||||
<class 'type | 436 | 443.60 KB
|
||||
<class 'code | 2974 | 419.56 KB
|
||||
<class '_io.BufferedWriter | 2 | 256.34 KB
|
||||
<class 'set | 420 | 159.88 KB
|
||||
<class '_io.BufferedReader | 1 | 128.17 KB
|
||||
<class 'wrapper_descriptor | 1130 | 88.28 KB
|
||||
<class 'tuple | 1304 | 86.57 KB
|
||||
<class 'weakref | 1013 | 79.14 KB
|
||||
<class 'builtin_function_or_method | 958 | 67.36 KB
|
||||
<class 'method_descriptor | 865 | 60.82 KB
|
||||
<class 'abc.ABCMeta | 62 | 59.96 KB
|
||||
<class 'list | 446 | 58.52 KB
|
||||
<class 'int | 1425 | 43.20 KB
|
||||
>>> from pympler import muppy
|
||||
>>> all_objects = muppy.get_objects()
|
||||
>>> len(all_objects)
|
||||
28667
|
||||
>>> from pympler import summary
|
||||
>>> suml = summary.summarize(all_objects)
|
||||
>>> summary.print_(suml)
|
||||
types | # objects | total size
|
||||
==================================== | =========== | ============
|
||||
<class 'str | 9822 | 1.10 MB
|
||||
<class 'dict | 1658 | 856.62 KB
|
||||
<class 'type | 436 | 443.60 KB
|
||||
<class 'code | 2974 | 419.56 KB
|
||||
<class '_io.BufferedWriter | 2 | 256.34 KB
|
||||
<class 'set | 420 | 159.88 KB
|
||||
<class '_io.BufferedReader | 1 | 128.17 KB
|
||||
<class 'wrapper_descriptor | 1130 | 88.28 KB
|
||||
<class 'tuple | 1304 | 86.57 KB
|
||||
<class 'weakref | 1013 | 79.14 KB
|
||||
<class 'builtin_function_or_method | 958 | 67.36 KB
|
||||
<class 'method_descriptor | 865 | 60.82 KB
|
||||
<class 'abc.ABCMeta | 62 | 59.96 KB
|
||||
<class 'list | 446 | 58.52 KB
|
||||
<class 'int | 1425 | 43.20 KB
|
||||
|
||||
For more info about muppy, refer to the `muppy documentation`_.
|
||||
|
||||
|
|
|
|||
|
|
@ -4,46 +4,33 @@
|
|||
Link Extractors
|
||||
===============
|
||||
|
||||
Link extractors are objects whose only purpose is to extract links from web
|
||||
pages (:class:`scrapy.http.Response` objects) which will be eventually
|
||||
followed.
|
||||
A link extractor is an object that extracts links from responses.
|
||||
|
||||
There is ``scrapy.linkextractors.LinkExtractor`` available
|
||||
in Scrapy, but you can create your own custom Link Extractors to suit your
|
||||
needs by implementing a simple interface.
|
||||
|
||||
The only public method that every link extractor has is ``extract_links``,
|
||||
which receives a :class:`~scrapy.http.Response` object and returns a list
|
||||
of :class:`scrapy.link.Link` objects. Link extractors are meant to be
|
||||
instantiated once and their ``extract_links`` method called several times
|
||||
with different responses to extract links to follow.
|
||||
|
||||
Link extractors are used in the :class:`~scrapy.spiders.CrawlSpider`
|
||||
class (available in Scrapy), through a set of rules, but you can also use it in
|
||||
your spiders, even if you don't subclass from
|
||||
:class:`~scrapy.spiders.CrawlSpider`, as its purpose is very simple: to
|
||||
extract links.
|
||||
The ``__init__`` method of
|
||||
:class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` takes settings that
|
||||
determine which links may be extracted. :class:`LxmlLinkExtractor.extract_links
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` returns a
|
||||
list of matching :class:`scrapy.link.Link` objects from a
|
||||
:class:`~scrapy.http.Response` object.
|
||||
|
||||
Link extractors are used in :class:`~scrapy.spiders.CrawlSpider` spiders
|
||||
through a set of :class:`~scrapy.spiders.Rule` objects. You can also use link
|
||||
extractors in regular spiders.
|
||||
|
||||
.. _topics-link-extractors-ref:
|
||||
|
||||
Built-in link extractors reference
|
||||
==================================
|
||||
Link extractor reference
|
||||
========================
|
||||
|
||||
.. module:: scrapy.linkextractors
|
||||
:synopsis: Link extractors classes
|
||||
|
||||
Link extractors classes bundled with Scrapy are provided in the
|
||||
:mod:`scrapy.linkextractors` module.
|
||||
|
||||
The default link extractor is ``LinkExtractor``, which is the same as
|
||||
:class:`~.LxmlLinkExtractor`::
|
||||
The link extractor class is
|
||||
:class:`scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`. For convenience it
|
||||
can also be imported as ``scrapy.linkextractors.LinkExtractor``::
|
||||
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
|
||||
There used to be other link extractor classes in previous Scrapy versions,
|
||||
but they are deprecated now.
|
||||
|
||||
LxmlLinkExtractor
|
||||
-----------------
|
||||
|
||||
|
|
@ -152,4 +139,6 @@ LxmlLinkExtractor
|
|||
from elements or attributes which allow leading/trailing whitespaces).
|
||||
:type strip: boolean
|
||||
|
||||
.. automethod:: extract_links
|
||||
|
||||
.. _scrapy.linkextractors: https://github.com/scrapy/scrapy/blob/master/scrapy/linkextractors/__init__.py
|
||||
|
|
|
|||
|
|
@ -206,14 +206,12 @@ metadata. Here is an example::
|
|||
output_processor=TakeFirst(),
|
||||
)
|
||||
|
||||
::
|
||||
|
||||
>>> from scrapy.loader import ItemLoader
|
||||
>>> il = ItemLoader(item=Product())
|
||||
>>> il.add_value('name', [u'Welcome to my', u'<strong>website</strong>'])
|
||||
>>> il.add_value('price', [u'€', u'<span>1000</span>'])
|
||||
>>> il.load_item()
|
||||
{'name': u'Welcome to my website', 'price': u'1000'}
|
||||
>>> from scrapy.loader import ItemLoader
|
||||
>>> il = ItemLoader(item=Product())
|
||||
>>> il.add_value('name', [u'Welcome to my', u'<strong>website</strong>'])
|
||||
>>> il.add_value('price', [u'€', u'<span>1000</span>'])
|
||||
>>> il.load_item()
|
||||
{'name': u'Welcome to my website', 'price': u'1000'}
|
||||
|
||||
The precedence order, for both input and output processors, is as follows:
|
||||
|
||||
|
|
@ -314,11 +312,11 @@ ItemLoader objects
|
|||
applied before processors
|
||||
:type re: str or compiled regex
|
||||
|
||||
Examples::
|
||||
Examples:
|
||||
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)')
|
||||
'FOO`
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)')
|
||||
'FOO`
|
||||
|
||||
.. method:: add_value(field_name, value, \*processors, \**kwargs)
|
||||
|
||||
|
|
@ -639,12 +637,12 @@ Here is a list of all built-in processors:
|
|||
values unchanged. It doesn't receive any ``__init__`` method arguments, nor does it
|
||||
accept Loader contexts.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import Identity
|
||||
>>> proc = Identity()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
['one', 'two', 'three']
|
||||
>>> from scrapy.loader.processors import Identity
|
||||
>>> proc = Identity()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
['one', 'two', 'three']
|
||||
|
||||
.. class:: TakeFirst
|
||||
|
||||
|
|
@ -652,12 +650,12 @@ Here is a list of all built-in processors:
|
|||
so it's typically used as an output processor to single-valued fields.
|
||||
It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> proc = TakeFirst()
|
||||
>>> proc(['', 'one', 'two', 'three'])
|
||||
'one'
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> proc = TakeFirst()
|
||||
>>> proc(['', 'one', 'two', 'three'])
|
||||
'one'
|
||||
|
||||
.. class:: Join(separator=u' ')
|
||||
|
||||
|
|
@ -667,15 +665,15 @@ Here is a list of all built-in processors:
|
|||
When using the default separator, this processor is equivalent to the
|
||||
function: ``u' '.join``
|
||||
|
||||
Examples::
|
||||
Examples:
|
||||
|
||||
>>> from scrapy.loader.processors import Join
|
||||
>>> proc = Join()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one two three'
|
||||
>>> proc = Join('<br>')
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one<br>two<br>three'
|
||||
>>> from scrapy.loader.processors import Join
|
||||
>>> proc = Join()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one two three'
|
||||
>>> proc = Join('<br>')
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one<br>two<br>three'
|
||||
|
||||
.. class:: Compose(\*functions, \**default_loader_context)
|
||||
|
||||
|
|
@ -688,12 +686,12 @@ Here is a list of all built-in processors:
|
|||
By default, stop process on ``None`` value. This behaviour can be changed by
|
||||
passing keyword argument ``stop_on_none=False``.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import Compose
|
||||
>>> proc = Compose(lambda v: v[0], str.upper)
|
||||
>>> proc(['hello', 'world'])
|
||||
'HELLO'
|
||||
>>> from scrapy.loader.processors import Compose
|
||||
>>> proc = Compose(lambda v: v[0], str.upper)
|
||||
>>> proc(['hello', 'world'])
|
||||
'HELLO'
|
||||
|
||||
Each function can optionally receive a ``loader_context`` parameter. For
|
||||
those which do, this processor will pass the currently active :ref:`Loader
|
||||
|
|
@ -732,15 +730,15 @@ Here is a list of all built-in processors:
|
|||
:meth:`~scrapy.selector.Selector.extract` method of :ref:`selectors
|
||||
<topics-selectors>`, which returns a list of unicode strings.
|
||||
|
||||
The example below should clarify how it works::
|
||||
The example below should clarify how it works:
|
||||
|
||||
>>> def filter_world(x):
|
||||
... return None if x == 'world' else x
|
||||
...
|
||||
>>> from scrapy.loader.processors import MapCompose
|
||||
>>> proc = MapCompose(filter_world, str.upper)
|
||||
>>> proc(['hello', 'world', 'this', 'is', 'scrapy'])
|
||||
['HELLO, 'THIS', 'IS', 'SCRAPY']
|
||||
>>> def filter_world(x):
|
||||
... return None if x == 'world' else x
|
||||
...
|
||||
>>> from scrapy.loader.processors import MapCompose
|
||||
>>> proc = MapCompose(filter_world, str.upper)
|
||||
>>> proc(['hello', 'world', 'this', 'is', 'scrapy'])
|
||||
['HELLO, 'THIS', 'IS', 'SCRAPY']
|
||||
|
||||
As with the Compose processor, functions can receive Loader contexts, and
|
||||
``__init__`` method keyword arguments are used as default context values. See
|
||||
|
|
@ -752,21 +750,21 @@ Here is a list of all built-in processors:
|
|||
Requires jmespath (https://github.com/jmespath/jmespath.py) to run.
|
||||
This processor takes only one input at a time.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose
|
||||
>>> proc = SelectJmes("foo") #for direct use on lists and dictionaries
|
||||
>>> proc({'foo': 'bar'})
|
||||
'bar'
|
||||
>>> proc({'foo': {'bar': 'baz'}})
|
||||
{'bar': 'baz'}
|
||||
>>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose
|
||||
>>> proc = SelectJmes("foo") #for direct use on lists and dictionaries
|
||||
>>> proc({'foo': 'bar'})
|
||||
'bar'
|
||||
>>> proc({'foo': {'bar': 'baz'}})
|
||||
{'bar': 'baz'}
|
||||
|
||||
Working with Json::
|
||||
Working with Json:
|
||||
|
||||
>>> import json
|
||||
>>> proc_single_json_str = Compose(json.loads, SelectJmes("foo"))
|
||||
>>> proc_single_json_str('{"foo": "bar"}')
|
||||
'bar'
|
||||
>>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo')))
|
||||
>>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]')
|
||||
['bar']
|
||||
>>> import json
|
||||
>>> proc_single_json_str = Compose(json.loads, SelectJmes("foo"))
|
||||
>>> proc_single_json_str('{"foo": "bar"}')
|
||||
'bar'
|
||||
>>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo')))
|
||||
>>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]')
|
||||
['bar']
|
||||
|
|
|
|||
|
|
@ -171,9 +171,9 @@ listed in `logging's logrecord attributes docs
|
|||
<https://docs.python.org/2/library/datetime.html#strftime-and-strptime-behavior>`_
|
||||
respectively.
|
||||
|
||||
If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the scrapy
|
||||
If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the Scrapy
|
||||
component that prints the log. It is unset by default, hence logs contain the
|
||||
scrapy component responsible for that log output.
|
||||
Scrapy component responsible for that log output.
|
||||
|
||||
Command-line options
|
||||
--------------------
|
||||
|
|
|
|||
|
|
@ -558,7 +558,7 @@ See here the methods that you can override in your custom Images Pipeline:
|
|||
Custom Images pipeline example
|
||||
==============================
|
||||
|
||||
Here is a full example of the Images Pipeline whose methods are examplified
|
||||
Here is a full example of the Images Pipeline whose methods are exemplified
|
||||
above::
|
||||
|
||||
import scrapy
|
||||
|
|
|
|||
|
|
@ -51,18 +51,18 @@ Constructing selectors
|
|||
.. highlight:: python
|
||||
|
||||
Response objects expose a :class:`~scrapy.selector.Selector` instance
|
||||
on ``.selector`` attribute::
|
||||
on ``.selector`` attribute:
|
||||
|
||||
>>> response.selector.xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> response.selector.xpath('//span/text()').get()
|
||||
'good'
|
||||
|
||||
Querying responses using XPath and CSS is so common that responses include two
|
||||
more shortcuts: ``response.xpath()`` and ``response.css()``::
|
||||
more shortcuts: ``response.xpath()`` and ``response.css()``:
|
||||
|
||||
>>> response.xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> response.css('span::text').get()
|
||||
'good'
|
||||
>>> response.xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> response.css('span::text').get()
|
||||
'good'
|
||||
|
||||
Scrapy selectors are instances of :class:`~scrapy.selector.Selector` class
|
||||
constructed by passing either :class:`~scrapy.http.TextResponse` object or
|
||||
|
|
@ -74,21 +74,21 @@ shortcuts. By using ``response.selector`` or one of these shortcuts
|
|||
you can also ensure the response body is parsed only once.
|
||||
|
||||
But if required, it is possible to use ``Selector`` directly.
|
||||
Constructing from text::
|
||||
Constructing from text:
|
||||
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> body = '<html><body><span>good</span></body></html>'
|
||||
>>> Selector(text=body).xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> body = '<html><body><span>good</span></body></html>'
|
||||
>>> Selector(text=body).xpath('//span/text()').get()
|
||||
'good'
|
||||
|
||||
Constructing from response - :class:`~scrapy.http.HtmlResponse` is one of
|
||||
:class:`~scrapy.http.TextResponse` subclasses::
|
||||
:class:`~scrapy.http.TextResponse` subclasses:
|
||||
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> from scrapy.http import HtmlResponse
|
||||
>>> response = HtmlResponse(url='http://example.com', body=body)
|
||||
>>> Selector(response=response).xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> from scrapy.http import HtmlResponse
|
||||
>>> response = HtmlResponse(url='http://example.com', body=body)
|
||||
>>> Selector(response=response).xpath('//span/text()').get()
|
||||
'good'
|
||||
|
||||
``Selector`` automatically chooses the best parsing rules
|
||||
(XML vs HTML) based on input type.
|
||||
|
|
@ -123,118 +123,118 @@ Since we're dealing with HTML, the selector will automatically use an HTML parse
|
|||
.. highlight:: python
|
||||
|
||||
So, by looking at the :ref:`HTML code <topics-selectors-htmlcode>` of that
|
||||
page, let's construct an XPath for selecting the text inside the title tag::
|
||||
page, let's construct an XPath for selecting the text inside the title tag:
|
||||
|
||||
>>> response.xpath('//title/text()')
|
||||
[<Selector xpath='//title/text()' data='Example website'>]
|
||||
>>> response.xpath('//title/text()')
|
||||
[<Selector xpath='//title/text()' data='Example website'>]
|
||||
|
||||
To actually extract the textual data, you must call the selector ``.get()``
|
||||
or ``.getall()`` methods, as follows::
|
||||
or ``.getall()`` methods, as follows:
|
||||
|
||||
>>> response.xpath('//title/text()').getall()
|
||||
['Example website']
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Example website'
|
||||
>>> response.xpath('//title/text()').getall()
|
||||
['Example website']
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Example website'
|
||||
|
||||
``.get()`` always returns a single result; if there are several matches,
|
||||
content of a first match is returned; if there are no matches, None
|
||||
is returned. ``.getall()`` returns a list with all results.
|
||||
|
||||
Notice that CSS selectors can select text or attribute nodes using CSS3
|
||||
pseudo-elements::
|
||||
pseudo-elements:
|
||||
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
|
||||
As you can see, ``.xpath()`` and ``.css()`` methods return a
|
||||
:class:`~scrapy.selector.SelectorList` instance, which is a list of new
|
||||
selectors. This API can be used for quickly selecting nested data::
|
||||
selectors. This API can be used for quickly selecting nested data:
|
||||
|
||||
>>> response.css('img').xpath('@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> response.css('img').xpath('@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
If you want to extract only the first matched element, you can call the
|
||||
selector ``.get()`` (or its alias ``.extract_first()`` commonly used in
|
||||
previous Scrapy versions)::
|
||||
previous Scrapy versions):
|
||||
|
||||
>>> response.xpath('//div[@id="images"]/a/text()').get()
|
||||
'Name: My image 1 '
|
||||
>>> response.xpath('//div[@id="images"]/a/text()').get()
|
||||
'Name: My image 1 '
|
||||
|
||||
It returns ``None`` if no element was found::
|
||||
It returns ``None`` if no element was found:
|
||||
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get() is None
|
||||
True
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get() is None
|
||||
True
|
||||
|
||||
A default return value can be provided as an argument, to be used instead
|
||||
of ``None``:
|
||||
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found')
|
||||
'not-found'
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found')
|
||||
'not-found'
|
||||
|
||||
Instead of using e.g. ``'@src'`` XPath it is possible to query for attributes
|
||||
using ``.attrib`` property of a :class:`~scrapy.selector.Selector`::
|
||||
using ``.attrib`` property of a :class:`~scrapy.selector.Selector`:
|
||||
|
||||
>>> [img.attrib['src'] for img in response.css('img')]
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> [img.attrib['src'] for img in response.css('img')]
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
As a shortcut, ``.attrib`` is also available on SelectorList directly;
|
||||
it returns attributes for the first matching element::
|
||||
it returns attributes for the first matching element:
|
||||
|
||||
>>> response.css('img').attrib['src']
|
||||
'image1_thumb.jpg'
|
||||
>>> response.css('img').attrib['src']
|
||||
'image1_thumb.jpg'
|
||||
|
||||
This is most useful when only a single result is expected, e.g. when selecting
|
||||
by id, or selecting unique elements on a web page::
|
||||
by id, or selecting unique elements on a web page:
|
||||
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
|
||||
Now we're going to get the base URL and some image links::
|
||||
Now we're going to get the base URL and some image links:
|
||||
|
||||
>>> response.xpath('//base/@href').get()
|
||||
'http://example.com/'
|
||||
>>> response.xpath('//base/@href').get()
|
||||
'http://example.com/'
|
||||
|
||||
>>> response.css('base::attr(href)').get()
|
||||
'http://example.com/'
|
||||
>>> response.css('base::attr(href)').get()
|
||||
'http://example.com/'
|
||||
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/@href').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
>>> response.xpath('//a[contains(@href, "image")]/@href').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
|
||||
>>> response.css('a[href*=image]::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
>>> response.css('a[href*=image]::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/img/@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> response.xpath('//a[contains(@href, "image")]/img/@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
>>> response.css('a[href*=image] img::attr(src)').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> response.css('a[href*=image] img::attr(src)').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
.. _topics-selectors-css-extensions:
|
||||
|
||||
|
|
@ -259,47 +259,47 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**:
|
|||
|
||||
Examples:
|
||||
|
||||
* ``title::text`` selects children text nodes of a descendant ``<title>`` element::
|
||||
* ``title::text`` selects children text nodes of a descendant ``<title>`` element:
|
||||
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
|
||||
* ``*::text`` selects all descendant text nodes of the current selector context::
|
||||
* ``*::text`` selects all descendant text nodes of the current selector context:
|
||||
|
||||
>>> response.css('#images *::text').getall()
|
||||
['\n ',
|
||||
'Name: My image 1 ',
|
||||
'\n ',
|
||||
'Name: My image 2 ',
|
||||
'\n ',
|
||||
'Name: My image 3 ',
|
||||
'\n ',
|
||||
'Name: My image 4 ',
|
||||
'\n ',
|
||||
'Name: My image 5 ',
|
||||
'\n ']
|
||||
>>> response.css('#images *::text').getall()
|
||||
['\n ',
|
||||
'Name: My image 1 ',
|
||||
'\n ',
|
||||
'Name: My image 2 ',
|
||||
'\n ',
|
||||
'Name: My image 3 ',
|
||||
'\n ',
|
||||
'Name: My image 4 ',
|
||||
'\n ',
|
||||
'Name: My image 5 ',
|
||||
'\n ']
|
||||
|
||||
* ``foo::text`` returns no results if ``foo`` element exists, but contains
|
||||
no text (i.e. text is empty)::
|
||||
no text (i.e. text is empty):
|
||||
|
||||
>>> response.css('img::text').getall()
|
||||
[]
|
||||
>>> response.css('img::text').getall()
|
||||
[]
|
||||
|
||||
This means ``.css('foo::text').get()`` could return None even if an element
|
||||
exists. Use ``default=''`` if you always want a string::
|
||||
exists. Use ``default=''`` if you always want a string:
|
||||
|
||||
>>> response.css('img::text').get()
|
||||
>>> response.css('img::text').get(default='')
|
||||
''
|
||||
>>> response.css('img::text').get()
|
||||
>>> response.css('img::text').get(default='')
|
||||
''
|
||||
|
||||
* ``a::attr(href)`` selects the *href* attribute value of descendant links::
|
||||
* ``a::attr(href)`` selects the *href* attribute value of descendant links:
|
||||
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
|
||||
.. note::
|
||||
See also: :ref:`selecting-attributes`.
|
||||
|
|
@ -318,25 +318,24 @@ Nesting selectors
|
|||
|
||||
The selection methods (``.xpath()`` or ``.css()``) return a list of selectors
|
||||
of the same type, so you can call the selection methods for those selectors
|
||||
too. Here's an example::
|
||||
too. Here's an example:
|
||||
|
||||
>>> links = response.xpath('//a[contains(@href, "image")]')
|
||||
>>> links.getall()
|
||||
['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>',
|
||||
'<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>',
|
||||
'<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>',
|
||||
'<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>',
|
||||
'<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>']
|
||||
>>> links = response.xpath('//a[contains(@href, "image")]')
|
||||
>>> links.getall()
|
||||
['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>',
|
||||
'<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>',
|
||||
'<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>',
|
||||
'<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>',
|
||||
'<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>']
|
||||
|
||||
>>> for index, link in enumerate(links):
|
||||
... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get())
|
||||
... print('Link number %d points to url %r and image %r' % args)
|
||||
|
||||
Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg'
|
||||
Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg'
|
||||
Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg'
|
||||
Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg'
|
||||
Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg'
|
||||
>>> for index, link in enumerate(links):
|
||||
... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get())
|
||||
... print('Link number %d points to url %r and image %r' % args)
|
||||
Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg'
|
||||
Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg'
|
||||
Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg'
|
||||
Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg'
|
||||
Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg'
|
||||
|
||||
.. _selecting-attributes:
|
||||
|
||||
|
|
@ -344,42 +343,42 @@ Selecting element attributes
|
|||
----------------------------
|
||||
|
||||
There are several ways to get a value of an attribute. First, one can use
|
||||
XPath syntax::
|
||||
XPath syntax:
|
||||
|
||||
>>> response.xpath("//a/@href").getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.xpath("//a/@href").getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
XPath syntax has a few advantages: it is a standard XPath feature, and
|
||||
``@attributes`` can be used in other parts of an XPath expression - e.g.
|
||||
it is possible to filter by attribute value.
|
||||
|
||||
Scrapy also provides an extension to CSS selectors (``::attr(...)``)
|
||||
which allows to get attribute values::
|
||||
which allows to get attribute values:
|
||||
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
In addition to that, there is a ``.attrib`` property of Selector.
|
||||
You can use it if you prefer to lookup attributes in Python
|
||||
code, without using XPaths or CSS extensions::
|
||||
code, without using XPaths or CSS extensions:
|
||||
|
||||
>>> [a.attrib['href'] for a in response.css('a')]
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> [a.attrib['href'] for a in response.css('a')]
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
This property is also available on SelectorList; it returns a dictionary
|
||||
with attributes of a first matching element. It is convenient to use when
|
||||
a selector is expected to give a single result (e.g. when selecting by element
|
||||
ID, or when selecting an unique element on a page)::
|
||||
ID, or when selecting an unique element on a page):
|
||||
|
||||
>>> response.css('base').attrib
|
||||
{'href': 'http://example.com/'}
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
>>> response.css('base').attrib
|
||||
{'href': 'http://example.com/'}
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
|
||||
``.attrib`` property of an empty SelectorList is empty::
|
||||
``.attrib`` property of an empty SelectorList is empty:
|
||||
|
||||
>>> response.css('foo').attrib
|
||||
{}
|
||||
>>> response.css('foo').attrib
|
||||
{}
|
||||
|
||||
Using selectors with regular expressions
|
||||
----------------------------------------
|
||||
|
|
@ -390,21 +389,21 @@ data using regular expressions. However, unlike using ``.xpath()`` or
|
|||
can't construct nested ``.re()`` calls.
|
||||
|
||||
Here's an example used to extract image names from the :ref:`HTML code
|
||||
<topics-selectors-htmlcode>` above::
|
||||
<topics-selectors-htmlcode>` above:
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)')
|
||||
['My image 1',
|
||||
'My image 2',
|
||||
'My image 3',
|
||||
'My image 4',
|
||||
'My image 5']
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)')
|
||||
['My image 1',
|
||||
'My image 2',
|
||||
'My image 3',
|
||||
'My image 4',
|
||||
'My image 5']
|
||||
|
||||
There's an additional helper reciprocating ``.get()`` (and its
|
||||
alias ``.extract_first()``) for ``.re()``, named ``.re_first()``.
|
||||
Use it to extract just the first matching string::
|
||||
Use it to extract just the first matching string:
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)')
|
||||
'My image 1'
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)')
|
||||
'My image 1'
|
||||
|
||||
.. _old-extraction-api:
|
||||
|
||||
|
|
@ -422,28 +421,28 @@ and readable code.
|
|||
|
||||
The following examples show how these methods map to each other.
|
||||
|
||||
1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``::
|
||||
1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``:
|
||||
|
||||
>>> response.css('a::attr(href)').get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)').extract_first()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)').get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)').extract_first()
|
||||
'image1.html'
|
||||
|
||||
2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``::
|
||||
2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``:
|
||||
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').extract()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').extract()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
3. ``Selector.get()`` is the same as ``Selector.extract()``::
|
||||
3. ``Selector.get()`` is the same as ``Selector.extract()``:
|
||||
|
||||
>>> response.css('a::attr(href)')[0].get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)')[0].extract()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)')[0].get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)')[0].extract()
|
||||
'image1.html'
|
||||
|
||||
4. For consistency, there is also ``Selector.getall()``, which returns a list::
|
||||
4. For consistency, there is also ``Selector.getall()``, which returns a list:
|
||||
|
||||
>>> response.css('a::attr(href)')[0].getall()
|
||||
['image1.html']
|
||||
|
|
@ -481,26 +480,26 @@ with ``/``, that XPath will be absolute to the document and not relative to the
|
|||
``Selector`` you're calling it from.
|
||||
|
||||
For example, suppose you want to extract all ``<p>`` elements inside ``<div>``
|
||||
elements. First, you would get all ``<div>`` elements::
|
||||
elements. First, you would get all ``<div>`` elements:
|
||||
|
||||
>>> divs = response.xpath('//div')
|
||||
>>> divs = response.xpath('//div')
|
||||
|
||||
At first, you may be tempted to use the following approach, which is wrong, as
|
||||
it actually extracts all ``<p>`` elements from the document, not only those
|
||||
inside ``<div>`` elements::
|
||||
inside ``<div>`` elements:
|
||||
|
||||
>>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document
|
||||
... print(p.get())
|
||||
>>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document
|
||||
... print(p.get())
|
||||
|
||||
This is the proper way to do it (note the dot prefixing the ``.//p`` XPath)::
|
||||
This is the proper way to do it (note the dot prefixing the ``.//p`` XPath):
|
||||
|
||||
>>> for p in divs.xpath('.//p'): # extracts all <p> inside
|
||||
... print(p.get())
|
||||
>>> for p in divs.xpath('.//p'): # extracts all <p> inside
|
||||
... print(p.get())
|
||||
|
||||
Another common case would be to extract all direct ``<p>`` children::
|
||||
Another common case would be to extract all direct ``<p>`` children:
|
||||
|
||||
>>> for p in divs.xpath('p'):
|
||||
... print(p.get())
|
||||
>>> for p in divs.xpath('p'):
|
||||
... print(p.get())
|
||||
|
||||
For more details about relative XPaths see the `Location Paths`_ section in the
|
||||
XPath specification.
|
||||
|
|
@ -521,12 +520,12 @@ for that you may end up with more elements that you want, if they have a differe
|
|||
class name that shares the string ``someclass``.
|
||||
|
||||
As it turns out, Scrapy selectors allow you to chain selectors, so most of the time
|
||||
you can just select by class using CSS and then switch to XPath when needed::
|
||||
you can just select by class using CSS and then switch to XPath when needed:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>')
|
||||
>>> sel.css('.shout').xpath('./time/@datetime').getall()
|
||||
['2014-07-23 19:00']
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>')
|
||||
>>> sel.css('.shout').xpath('./time/@datetime').getall()
|
||||
['2014-07-23 19:00']
|
||||
|
||||
This is cleaner than using the verbose XPath trick shown above. Just remember
|
||||
to use the ``.`` in the XPath expressions that will follow.
|
||||
|
|
@ -538,41 +537,41 @@ Beware of the difference between //node[1] and (//node)[1]
|
|||
|
||||
``(//node)[1]`` selects all the nodes in the document, and then gets only the first of them.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text="""
|
||||
....: <ul class="list">
|
||||
....: <li>1</li>
|
||||
....: <li>2</li>
|
||||
....: <li>3</li>
|
||||
....: </ul>
|
||||
....: <ul class="list">
|
||||
....: <li>4</li>
|
||||
....: <li>5</li>
|
||||
....: <li>6</li>
|
||||
....: </ul>""")
|
||||
>>> xp = lambda x: sel.xpath(x).getall()
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text="""
|
||||
....: <ul class="list">
|
||||
....: <li>1</li>
|
||||
....: <li>2</li>
|
||||
....: <li>3</li>
|
||||
....: </ul>
|
||||
....: <ul class="list">
|
||||
....: <li>4</li>
|
||||
....: <li>5</li>
|
||||
....: <li>6</li>
|
||||
....: </ul>""")
|
||||
>>> xp = lambda x: sel.xpath(x).getall()
|
||||
|
||||
This gets all first ``<li>`` elements under whatever it is its parent::
|
||||
This gets all first ``<li>`` elements under whatever it is its parent:
|
||||
|
||||
>>> xp("//li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
>>> xp("//li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
|
||||
And this gets the first ``<li>`` element in the whole document::
|
||||
And this gets the first ``<li>`` element in the whole document:
|
||||
|
||||
>>> xp("(//li)[1]")
|
||||
['<li>1</li>']
|
||||
>>> xp("(//li)[1]")
|
||||
['<li>1</li>']
|
||||
|
||||
This gets all first ``<li>`` elements under an ``<ul>`` parent::
|
||||
This gets all first ``<li>`` elements under an ``<ul>`` parent:
|
||||
|
||||
>>> xp("//ul/li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
>>> xp("//ul/li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
|
||||
And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document::
|
||||
And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document:
|
||||
|
||||
>>> xp("(//ul/li)[1]")
|
||||
['<li>1</li>']
|
||||
>>> xp("(//ul/li)[1]")
|
||||
['<li>1</li>']
|
||||
|
||||
Using text nodes in a condition
|
||||
-------------------------------
|
||||
|
|
@ -584,34 +583,34 @@ This is because the expression ``.//text()`` yields a collection of text element
|
|||
And when a node-set is converted to a string, which happens when it is passed as argument to
|
||||
a string function like ``contains()`` or ``starts-with()``, it results in the text for the first element only.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>')
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>')
|
||||
|
||||
Converting a *node-set* to string::
|
||||
Converting a *node-set* to string:
|
||||
|
||||
>>> sel.xpath('//a//text()').getall() # take a peek at the node-set
|
||||
['Click here to go to the ', 'Next Page']
|
||||
>>> sel.xpath("string(//a[1]//text())").getall() # convert it to string
|
||||
['Click here to go to the ']
|
||||
>>> sel.xpath('//a//text()').getall() # take a peek at the node-set
|
||||
['Click here to go to the ', 'Next Page']
|
||||
>>> sel.xpath("string(//a[1]//text())").getall() # convert it to string
|
||||
['Click here to go to the ']
|
||||
|
||||
A *node* converted to a string, however, puts together the text of itself plus of all its descendants::
|
||||
A *node* converted to a string, however, puts together the text of itself plus of all its descendants:
|
||||
|
||||
>>> sel.xpath("//a[1]").getall() # select the first node
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
>>> sel.xpath("string(//a[1])").getall() # convert it to string
|
||||
['Click here to go to the Next Page']
|
||||
>>> sel.xpath("//a[1]").getall() # select the first node
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
>>> sel.xpath("string(//a[1])").getall() # convert it to string
|
||||
['Click here to go to the Next Page']
|
||||
|
||||
So, using the ``.//text()`` node-set won't select anything in this case::
|
||||
So, using the ``.//text()`` node-set won't select anything in this case:
|
||||
|
||||
>>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall()
|
||||
[]
|
||||
>>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall()
|
||||
[]
|
||||
|
||||
But using the ``.`` to mean the node, works::
|
||||
But using the ``.`` to mean the node, works:
|
||||
|
||||
>>> sel.xpath("//a[contains(., 'Next Page')]").getall()
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
>>> sel.xpath("//a[contains(., 'Next Page')]").getall()
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
|
||||
.. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions
|
||||
|
||||
|
|
@ -627,17 +626,17 @@ some arguments in your queries with placeholders like ``?``,
|
|||
which are then substituted with values passed with the query.
|
||||
|
||||
Here's an example to match an element based on its "id" attribute value,
|
||||
without hard-coding it (that was shown previously)::
|
||||
without hard-coding it (that was shown previously):
|
||||
|
||||
>>> # `$val` used in the expression, a `val` argument needs to be passed
|
||||
>>> response.xpath('//div[@id=$val]/a/text()', val='images').get()
|
||||
'Name: My image 1 '
|
||||
>>> # `$val` used in the expression, a `val` argument needs to be passed
|
||||
>>> response.xpath('//div[@id=$val]/a/text()', val='images').get()
|
||||
'Name: My image 1 '
|
||||
|
||||
Here's another example, to find the "id" attribute of a ``<div>`` tag containing
|
||||
five ``<a>`` children (here we pass the value ``5`` as an integer)::
|
||||
five ``<a>`` children (here we pass the value ``5`` as an integer):
|
||||
|
||||
>>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get()
|
||||
'images'
|
||||
>>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get()
|
||||
'images'
|
||||
|
||||
All variable references must have a binding value when calling ``.xpath()``
|
||||
(otherwise you'll get a ``ValueError: XPath error:`` exception).
|
||||
|
|
@ -687,19 +686,19 @@ You can see several namespace declarations including a default
|
|||
.. highlight:: python
|
||||
|
||||
Once in the shell we can try selecting all ``<link>`` objects and see that it
|
||||
doesn't work (because the Atom XML namespace is obfuscating those nodes)::
|
||||
doesn't work (because the Atom XML namespace is obfuscating those nodes):
|
||||
|
||||
>>> response.xpath("//link")
|
||||
[]
|
||||
>>> response.xpath("//link")
|
||||
[]
|
||||
|
||||
But once we call the :meth:`Selector.remove_namespaces` method, all
|
||||
nodes can be accessed directly by their names::
|
||||
nodes can be accessed directly by their names:
|
||||
|
||||
>>> response.selector.remove_namespaces()
|
||||
>>> response.xpath("//link")
|
||||
[<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>,
|
||||
<Selector xpath='//link' data='<link rel="next" type="application/atom+'>,
|
||||
...
|
||||
>>> response.selector.remove_namespaces()
|
||||
>>> response.xpath("//link")
|
||||
[<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>,
|
||||
<Selector xpath='//link' data='<link rel="next" type="application/atom+'>,
|
||||
...
|
||||
|
||||
If you wonder why the namespace removal procedure isn't always called by default
|
||||
instead of having to call it manually, this is because of two reasons, which, in order
|
||||
|
|
@ -734,26 +733,25 @@ Regular expressions
|
|||
The ``test()`` function, for example, can prove quite useful when XPath's
|
||||
``starts-with()`` or ``contains()`` are not sufficient.
|
||||
|
||||
Example selecting links in list item with a "class" attribute ending with a digit::
|
||||
Example selecting links in list item with a "class" attribute ending with a digit:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> doc = u"""
|
||||
... <div>
|
||||
... <ul>
|
||||
... <li class="item-0"><a href="link1.html">first item</a></li>
|
||||
... <li class="item-1"><a href="link2.html">second item</a></li>
|
||||
... <li class="item-inactive"><a href="link3.html">third item</a></li>
|
||||
... <li class="item-1"><a href="link4.html">fourth item</a></li>
|
||||
... <li class="item-0"><a href="link5.html">fifth item</a></li>
|
||||
... </ul>
|
||||
... </div>
|
||||
... """
|
||||
>>> sel = Selector(text=doc, type="html")
|
||||
>>> sel.xpath('//li//@href').getall()
|
||||
['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html']
|
||||
>>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall()
|
||||
['link1.html', 'link2.html', 'link4.html', 'link5.html']
|
||||
>>>
|
||||
>>> from scrapy import Selector
|
||||
>>> doc = u"""
|
||||
... <div>
|
||||
... <ul>
|
||||
... <li class="item-0"><a href="link1.html">first item</a></li>
|
||||
... <li class="item-1"><a href="link2.html">second item</a></li>
|
||||
... <li class="item-inactive"><a href="link3.html">third item</a></li>
|
||||
... <li class="item-1"><a href="link4.html">fourth item</a></li>
|
||||
... <li class="item-0"><a href="link5.html">fifth item</a></li>
|
||||
... </ul>
|
||||
... </div>
|
||||
... """
|
||||
>>> sel = Selector(text=doc, type="html")
|
||||
>>> sel.xpath('//li//@href').getall()
|
||||
['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html']
|
||||
>>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall()
|
||||
['link1.html', 'link2.html', 'link4.html', 'link5.html']
|
||||
|
||||
.. warning:: C library ``libxslt`` doesn't natively support EXSLT regular
|
||||
expressions so `lxml`_'s implementation uses hooks to Python's ``re`` module.
|
||||
|
|
@ -849,7 +847,6 @@ with groups of itemscopes and corresponding itemprops::
|
|||
current scope: ['http://schema.org/Rating']
|
||||
properties: ['worstRating', 'ratingValue', 'bestRating']
|
||||
|
||||
>>>
|
||||
|
||||
Here we first iterate over ``itemscope`` elements, and for each one,
|
||||
we look for all ``itemprops`` elements and exclude those that are themselves
|
||||
|
|
@ -877,15 +874,15 @@ For the following HTML::
|
|||
|
||||
.. highlight:: python
|
||||
|
||||
You can use it like this::
|
||||
You can use it like this:
|
||||
|
||||
>>> response.xpath('//p[has-class("foo")]')
|
||||
[<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>,
|
||||
<Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar-baz")]')
|
||||
[<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar")]')
|
||||
[]
|
||||
>>> response.xpath('//p[has-class("foo")]')
|
||||
[<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>,
|
||||
<Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar-baz")]')
|
||||
[<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar")]')
|
||||
[]
|
||||
|
||||
So XPath ``//p[has-class("foo", "bar-baz")]`` is roughly equivalent to CSS
|
||||
``p.foo.bar-baz``. Please note, that it is slower in most of the cases,
|
||||
|
|
|
|||
|
|
@ -909,7 +909,7 @@ LOG_FORMAT
|
|||
|
||||
Default: ``'%(asctime)s [%(name)s] %(levelname)s: %(message)s'``
|
||||
|
||||
String for formatting log messsages. Refer to the `Python logging documentation`_ for the whole list of available
|
||||
String for formatting log messages. Refer to the `Python logging documentation`_ for the whole list of available
|
||||
placeholders.
|
||||
|
||||
.. _Python logging documentation: https://docs.python.org/2/library/logging.html#logrecord-attributes
|
||||
|
|
@ -1289,7 +1289,7 @@ Default::
|
|||
'scrapy.contracts.default.ScrapesContract': 3,
|
||||
}
|
||||
|
||||
A dict containing the scrapy contracts enabled by default in Scrapy. You should
|
||||
A dict containing the Scrapy contracts enabled by default in Scrapy. You should
|
||||
never modify this setting in your project, modify :setting:`SPIDER_CONTRACTS`
|
||||
instead. For more info see :ref:`topics-contracts`.
|
||||
|
||||
|
|
@ -1320,7 +1320,7 @@ SPIDER_LOADER_WARN_ONLY
|
|||
|
||||
Default: ``False``
|
||||
|
||||
By default, when scrapy tries to import spider classes from :setting:`SPIDER_MODULES`,
|
||||
By default, when Scrapy tries to import spider classes from :setting:`SPIDER_MODULES`,
|
||||
it will fail loudly if there is any ``ImportError`` exception.
|
||||
But you can choose to silence this exception and turn it into a simple
|
||||
warning by setting ``SPIDER_LOADER_WARN_ONLY = True``.
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ for more info.
|
|||
Scrapy also has support for `bpython`_, and will try to use it where `IPython`_
|
||||
is unavailable.
|
||||
|
||||
Through scrapy's settings you can configure it to use any one of
|
||||
Through Scrapy's settings you can configure it to use any one of
|
||||
``ipython``, ``bpython`` or the standard ``python`` shell, regardless of which
|
||||
are installed. This is done by setting the ``SCRAPY_PYTHON_SHELL`` environment
|
||||
variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`::
|
||||
|
|
@ -177,47 +177,46 @@ all start with the ``[s]`` prefix)::
|
|||
>>>
|
||||
|
||||
|
||||
After that, we can start playing with the objects::
|
||||
After that, we can start playing with the objects:
|
||||
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
|
||||
|
||||
>>> fetch("https://reddit.com")
|
||||
>>> fetch("https://reddit.com")
|
||||
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'reddit: the front page of the internet'
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'reddit: the front page of the internet'
|
||||
|
||||
>>> request = request.replace(method="POST")
|
||||
>>> request = request.replace(method="POST")
|
||||
|
||||
>>> fetch(request)
|
||||
>>> fetch(request)
|
||||
|
||||
>>> response.status
|
||||
404
|
||||
>>> response.status
|
||||
404
|
||||
|
||||
>>> from pprint import pprint
|
||||
>>> from pprint import pprint
|
||||
|
||||
>>> pprint(response.headers)
|
||||
{'Accept-Ranges': ['bytes'],
|
||||
'Cache-Control': ['max-age=0, must-revalidate'],
|
||||
'Content-Type': ['text/html; charset=UTF-8'],
|
||||
'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'],
|
||||
'Server': ['snooserv'],
|
||||
'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'],
|
||||
'Vary': ['accept-encoding'],
|
||||
'Via': ['1.1 varnish'],
|
||||
'X-Cache': ['MISS'],
|
||||
'X-Cache-Hits': ['0'],
|
||||
'X-Content-Type-Options': ['nosniff'],
|
||||
'X-Frame-Options': ['SAMEORIGIN'],
|
||||
'X-Moose': ['majestic'],
|
||||
'X-Served-By': ['cache-cdg8730-CDG'],
|
||||
'X-Timer': ['S1481214079.394283,VS0,VE159'],
|
||||
'X-Ua-Compatible': ['IE=edge'],
|
||||
'X-Xss-Protection': ['1; mode=block']}
|
||||
>>>
|
||||
>>> pprint(response.headers)
|
||||
{'Accept-Ranges': ['bytes'],
|
||||
'Cache-Control': ['max-age=0, must-revalidate'],
|
||||
'Content-Type': ['text/html; charset=UTF-8'],
|
||||
'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'],
|
||||
'Server': ['snooserv'],
|
||||
'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'],
|
||||
'Vary': ['accept-encoding'],
|
||||
'Via': ['1.1 varnish'],
|
||||
'X-Cache': ['MISS'],
|
||||
'X-Cache-Hits': ['0'],
|
||||
'X-Content-Type-Options': ['nosniff'],
|
||||
'X-Frame-Options': ['SAMEORIGIN'],
|
||||
'X-Moose': ['majestic'],
|
||||
'X-Served-By': ['cache-cdg8730-CDG'],
|
||||
'X-Timer': ['S1481214079.394283,VS0,VE159'],
|
||||
'X-Ua-Compatible': ['IE=edge'],
|
||||
'X-Xss-Protection': ['1; mode=block']}
|
||||
|
||||
|
||||
.. _topics-shell-inspect-response:
|
||||
|
|
@ -263,16 +262,16 @@ When you run the spider, you will get something similar to this::
|
|||
>>> response.url
|
||||
'http://example.org'
|
||||
|
||||
Then, you can check if the extraction code is working::
|
||||
Then, you can check if the extraction code is working:
|
||||
|
||||
>>> response.xpath('//h1[@class="fn"]')
|
||||
[]
|
||||
>>> response.xpath('//h1[@class="fn"]')
|
||||
[]
|
||||
|
||||
Nope, it doesn't. So you can open the response in your web browser and see if
|
||||
it's the response you were expecting::
|
||||
it's the response you were expecting:
|
||||
|
||||
>>> view(response)
|
||||
True
|
||||
>>> view(response)
|
||||
True
|
||||
|
||||
Finally you hit Ctrl-D (or Ctrl-Z in Windows) to exit the shell and resume the
|
||||
crawling::
|
||||
|
|
|
|||
|
|
@ -414,6 +414,12 @@ Crawling rules
|
|||
from which the request originated as second argument. It must return a
|
||||
``Request`` object or ``None`` (to filter out the request).
|
||||
|
||||
``errback`` is a callable or a string (in which case a method from the spider
|
||||
object with that name will be used) to be called if any exception is
|
||||
raised while processing a request generated by the rule.
|
||||
It receives a :class:`Twisted Failure <twisted.python.failure.Failure>`
|
||||
instance as first parameter.
|
||||
|
||||
CrawlSpider example
|
||||
~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
|
|
|
|||
|
|
@ -57,15 +57,15 @@ Set stat value only if lower than previous::
|
|||
|
||||
stats.min_value('min_free_memory_percent', value)
|
||||
|
||||
Get stat value::
|
||||
Get stat value:
|
||||
|
||||
>>> stats.get_value('custom_count')
|
||||
1
|
||||
>>> stats.get_value('custom_count')
|
||||
1
|
||||
|
||||
Get all stats::
|
||||
Get all stats:
|
||||
|
||||
>>> stats.get_stats()
|
||||
{'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)}
|
||||
>>> stats.get_stats()
|
||||
{'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)}
|
||||
|
||||
Available Stats Collectors
|
||||
==========================
|
||||
|
|
|
|||
|
|
@ -44,11 +44,11 @@ the console you need to type::
|
|||
>>>
|
||||
|
||||
By default Username is ``scrapy`` and Password is autogenerated. The
|
||||
autogenerated Password can be seen on scrapy logs like the example below::
|
||||
autogenerated Password can be seen on Scrapy logs like the example below::
|
||||
|
||||
2018-10-16 14:35:21 [scrapy.extensions.telnet] INFO: Telnet Password: 16f92501e8a59326
|
||||
|
||||
Default Username and Password can be overriden by the settings
|
||||
Default Username and Password can be overridden by the settings
|
||||
:setting:`TELNETCONSOLE_USERNAME` and :setting:`TELNETCONSOLE_PASSWORD`.
|
||||
|
||||
.. warning::
|
||||
|
|
|
|||
91
pytest.ini
91
pytest.ini
|
|
@ -8,7 +8,6 @@ addopts =
|
|||
--ignore=docs/_ext
|
||||
--ignore=docs/conf.py
|
||||
--ignore=docs/news.rst
|
||||
--ignore=docs/topics/developer-tools.rst
|
||||
--ignore=docs/topics/dynamic-content.rst
|
||||
--ignore=docs/topics/items.rst
|
||||
--ignore=docs/topics/leaks.rst
|
||||
|
|
@ -28,52 +27,52 @@ flake8-ignore =
|
|||
scrapy/http/__init__.py F401
|
||||
# Issues pending a review:
|
||||
# extras
|
||||
extras/qps-bench-server.py E261 E501
|
||||
extras/qpsclient.py E501 E261 E501
|
||||
extras/qps-bench-server.py E501
|
||||
extras/qpsclient.py E501 E501
|
||||
# scrapy/commands
|
||||
scrapy/commands/__init__.py E128 E501
|
||||
scrapy/commands/check.py E501
|
||||
scrapy/commands/crawl.py E501
|
||||
scrapy/commands/edit.py E501
|
||||
scrapy/commands/fetch.py E401 E501 E128 E502 E731
|
||||
scrapy/commands/fetch.py E401 E501 E128 E731
|
||||
scrapy/commands/genspider.py E128 E501 E502
|
||||
scrapy/commands/parse.py E128 E501 E731 E226
|
||||
scrapy/commands/runspider.py E501
|
||||
scrapy/commands/settings.py E128
|
||||
scrapy/commands/shell.py E128 E501 E502
|
||||
scrapy/commands/startproject.py E502 E127 E501 E128
|
||||
scrapy/commands/startproject.py E127 E501 E128
|
||||
scrapy/commands/version.py E501 E128
|
||||
# scrapy/contracts
|
||||
scrapy/contracts/__init__.py E501 W504
|
||||
scrapy/contracts/default.py E502 E128
|
||||
scrapy/contracts/default.py E128
|
||||
# scrapy/core
|
||||
scrapy/core/engine.py E261 E501 E128 E127 E306 E502
|
||||
scrapy/core/engine.py E501 E128 E127 E306 E502
|
||||
scrapy/core/scheduler.py E501
|
||||
scrapy/core/scraper.py E501 E306 E261 E128 W504
|
||||
scrapy/core/spidermw.py E501 E731 E502 E126 E226
|
||||
scrapy/core/scraper.py E501 E306 E128 W504
|
||||
scrapy/core/spidermw.py E501 E731 E126 E226
|
||||
scrapy/core/downloader/__init__.py E501
|
||||
scrapy/core/downloader/contextfactory.py E501 E128 E126
|
||||
scrapy/core/downloader/middleware.py E501 E502
|
||||
scrapy/core/downloader/tls.py E501 E305 E241
|
||||
scrapy/core/downloader/webclient.py E731 E501 E261 E502 E128 E126 E226
|
||||
scrapy/core/downloader/webclient.py E731 E501 E128 E126 E226
|
||||
scrapy/core/downloader/handlers/__init__.py E501
|
||||
scrapy/core/downloader/handlers/ftp.py E501 E305 E128 E127
|
||||
scrapy/core/downloader/handlers/http10.py E501
|
||||
scrapy/core/downloader/handlers/http11.py E501
|
||||
scrapy/core/downloader/handlers/s3.py E501 E502 E128 E126
|
||||
scrapy/core/downloader/handlers/s3.py E501 E128 E126
|
||||
# scrapy/downloadermiddlewares
|
||||
scrapy/downloadermiddlewares/ajaxcrawl.py E501 E226
|
||||
scrapy/downloadermiddlewares/decompression.py E501
|
||||
scrapy/downloadermiddlewares/defaultheaders.py E501
|
||||
scrapy/downloadermiddlewares/httpcache.py E501 E126
|
||||
scrapy/downloadermiddlewares/httpcompression.py E502 E128
|
||||
scrapy/downloadermiddlewares/httpcompression.py E501 E128
|
||||
scrapy/downloadermiddlewares/httpproxy.py E501
|
||||
scrapy/downloadermiddlewares/redirect.py E501 W504
|
||||
scrapy/downloadermiddlewares/retry.py E501 E126
|
||||
scrapy/downloadermiddlewares/robotstxt.py E501
|
||||
scrapy/downloadermiddlewares/stats.py E501
|
||||
# scrapy/extensions
|
||||
scrapy/extensions/closespider.py E501 E502 E128 E123
|
||||
scrapy/extensions/closespider.py E501 E128 E123
|
||||
scrapy/extensions/corestats.py E501
|
||||
scrapy/extensions/feedexport.py E128 E501
|
||||
scrapy/extensions/httpcache.py E128 E501 E303
|
||||
|
|
@ -87,13 +86,13 @@ flake8-ignore =
|
|||
scrapy/http/request/__init__.py E501
|
||||
scrapy/http/request/form.py E501 E123
|
||||
scrapy/http/request/json_request.py E501
|
||||
scrapy/http/response/__init__.py E501 E128 W293 W291
|
||||
scrapy/http/response/text.py E501 W293 E128 E124
|
||||
scrapy/http/response/__init__.py E501 E128
|
||||
scrapy/http/response/text.py E501 E128 E124
|
||||
# scrapy/linkextractors
|
||||
scrapy/linkextractors/__init__.py E731 E502 E501 E402
|
||||
scrapy/linkextractors/__init__.py E731 E501 E402 W504
|
||||
scrapy/linkextractors/lxmlhtml.py E501 E731 E226
|
||||
# scrapy/loader
|
||||
scrapy/loader/__init__.py E501 E502 E128
|
||||
scrapy/loader/__init__.py E501 E128
|
||||
scrapy/loader/processors.py E501
|
||||
# scrapy/pipelines
|
||||
scrapy/pipelines/files.py E116 E501 E266
|
||||
|
|
@ -104,7 +103,7 @@ flake8-ignore =
|
|||
scrapy/selector/unified.py E501 E111
|
||||
# scrapy/settings
|
||||
scrapy/settings/__init__.py E501
|
||||
scrapy/settings/default_settings.py E501 E261 E114 E116 E226
|
||||
scrapy/settings/default_settings.py E501 E114 E116 E226
|
||||
scrapy/settings/deprecated.py E501
|
||||
# scrapy/spidermiddlewares
|
||||
scrapy/spidermiddlewares/httperror.py E501
|
||||
|
|
@ -114,26 +113,25 @@ flake8-ignore =
|
|||
# scrapy/spiders
|
||||
scrapy/spiders/__init__.py E501 E402
|
||||
scrapy/spiders/crawl.py E501
|
||||
scrapy/spiders/feed.py E501 E261
|
||||
scrapy/spiders/feed.py E501
|
||||
scrapy/spiders/sitemap.py E501
|
||||
# scrapy/utils
|
||||
scrapy/utils/asyncio.py E501
|
||||
scrapy/utils/benchserver.py E501
|
||||
scrapy/utils/conf.py E402 E502 E501
|
||||
scrapy/utils/console.py E261 E306 E305
|
||||
scrapy/utils/conf.py E402 E501
|
||||
scrapy/utils/console.py E306 E305
|
||||
scrapy/utils/datatypes.py E501 E226
|
||||
scrapy/utils/decorators.py E501
|
||||
scrapy/utils/defer.py E501 E128
|
||||
scrapy/utils/deprecate.py E128 E501 E127 E502
|
||||
scrapy/utils/engine.py E261
|
||||
scrapy/utils/gz.py E305 E501 W504
|
||||
scrapy/utils/http.py F403 E226
|
||||
scrapy/utils/httpobj.py E501
|
||||
scrapy/utils/iterators.py E501 E701
|
||||
scrapy/utils/log.py E128 W503
|
||||
scrapy/utils/markup.py F403 W292
|
||||
scrapy/utils/markup.py F403
|
||||
scrapy/utils/misc.py E501 E226
|
||||
scrapy/utils/multipart.py F403 W292
|
||||
scrapy/utils/multipart.py F403
|
||||
scrapy/utils/project.py E501
|
||||
scrapy/utils/python.py E501
|
||||
scrapy/utils/reactor.py E226
|
||||
|
|
@ -148,18 +146,17 @@ flake8-ignore =
|
|||
scrapy/utils/url.py E501 F403 E128 F405
|
||||
# scrapy
|
||||
scrapy/__init__.py E402 E501
|
||||
scrapy/_monkeypatches.py W293
|
||||
scrapy/cmdline.py E502 E501
|
||||
scrapy/cmdline.py E501
|
||||
scrapy/crawler.py E501
|
||||
scrapy/dupefilters.py E501 E202
|
||||
scrapy/exceptions.py E501
|
||||
scrapy/exporters.py E501 E261 E226
|
||||
scrapy/exporters.py E501 E226
|
||||
scrapy/interfaces.py E501
|
||||
scrapy/item.py E501 E128
|
||||
scrapy/link.py E501
|
||||
scrapy/logformatter.py E501 W293
|
||||
scrapy/logformatter.py E501
|
||||
scrapy/mail.py E402 E128 E501 E502
|
||||
scrapy/middleware.py E502 E128 E501
|
||||
scrapy/middleware.py E128 E501
|
||||
scrapy/pqueues.py E501
|
||||
scrapy/responsetypes.py E128 E501 E305
|
||||
scrapy/robotstxt.py E501
|
||||
|
|
@ -174,15 +171,15 @@ flake8-ignore =
|
|||
tests/pipelines.py F841 E226
|
||||
tests/spiders.py E501 E127
|
||||
tests/test_closespider.py E501 E127
|
||||
tests/test_command_fetch.py E501 E261
|
||||
tests/test_command_fetch.py E501
|
||||
tests/test_command_parse.py E501 E128 E303 E226
|
||||
tests/test_command_shell.py E501 E128
|
||||
tests/test_commands.py E128 E501
|
||||
tests/test_contracts.py E501 E128 W293
|
||||
tests/test_contracts.py E501 E128
|
||||
tests/test_crawl.py E501 E741 E265
|
||||
tests/test_crawler.py F841 E306 E501
|
||||
tests/test_dependencies.py F841 E501 E305
|
||||
tests/test_downloader_handlers.py E124 E127 E128 E225 E261 E265 E501 E502 E701 E126 E226 E123
|
||||
tests/test_downloader_handlers.py E124 E127 E128 E225 E265 E501 E701 E126 E226 E123
|
||||
tests/test_downloadermiddleware.py E501
|
||||
tests/test_downloadermiddleware_ajaxcrawlable.py E501
|
||||
tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E303 E265 E126
|
||||
|
|
@ -193,18 +190,18 @@ flake8-ignore =
|
|||
tests/test_downloadermiddleware_httpcompression.py E501 E251 E126 E123
|
||||
tests/test_downloadermiddleware_httpproxy.py E501 E128
|
||||
tests/test_downloadermiddleware_redirect.py E501 E303 E128 E306 E127 E305
|
||||
tests/test_downloadermiddleware_retry.py E501 E128 W293 E251 E502 E303 E126
|
||||
tests/test_downloadermiddleware_retry.py E501 E128 E251 E303 E126
|
||||
tests/test_downloadermiddleware_robotstxt.py E501
|
||||
tests/test_downloadermiddleware_stats.py E501
|
||||
tests/test_dupefilters.py E221 E501 E741 W293 W291 E128 E124
|
||||
tests/test_engine.py E401 E501 E502 E128 E261
|
||||
tests/test_dupefilters.py E221 E501 E741 E128 E124
|
||||
tests/test_engine.py E401 E501 E128
|
||||
tests/test_exporters.py E501 E731 E306 E128 E124
|
||||
tests/test_extension_telnet.py F841
|
||||
tests/test_feedexport.py E501 F841 E241
|
||||
tests/test_http_cookies.py E501
|
||||
tests/test_http_headers.py E501
|
||||
tests/test_http_request.py E402 E501 E261 E127 E128 W293 E502 E128 E502 E126 E123
|
||||
tests/test_http_response.py E501 E301 E502 E128 E265
|
||||
tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123
|
||||
tests/test_http_response.py E501 E301 E128 E265
|
||||
tests/test_item.py E701 E128 F841 E306
|
||||
tests/test_link.py E501
|
||||
tests/test_linkextractors.py E501 E128 E124
|
||||
|
|
@ -213,29 +210,29 @@ flake8-ignore =
|
|||
tests/test_mail.py E128 E501 E305
|
||||
tests/test_middleware.py E501 E128
|
||||
tests/test_pipeline_crawl.py E131 E501 E128 E126
|
||||
tests/test_pipeline_files.py E501 W293 E303 E272 E226
|
||||
tests/test_pipeline_files.py E501 E303 E272 E226
|
||||
tests/test_pipeline_images.py F841 E501 E303
|
||||
tests/test_pipeline_media.py E501 E741 E731 E128 E261 E306 E502
|
||||
tests/test_pipeline_media.py E501 E741 E731 E128 E306 E502
|
||||
tests/test_proxy_connect.py E501 E741
|
||||
tests/test_request_cb_kwargs.py E501
|
||||
tests/test_responsetypes.py E501 E305
|
||||
tests/test_robotstxt_interface.py E501 W291 E501
|
||||
tests/test_robotstxt_interface.py E501 E501
|
||||
tests/test_scheduler.py E501 E126 E123
|
||||
tests/test_selector.py E501 E127
|
||||
tests/test_spider.py E501
|
||||
tests/test_spidermiddleware.py E501 E226
|
||||
tests/test_spidermiddleware_httperror.py E128 E501 E127 E121
|
||||
tests/test_spidermiddleware_offsite.py E501 E128 E111 W293
|
||||
tests/test_spidermiddleware_output_chain.py E501 W293 E226
|
||||
tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E261 E124 E501 E241 E121
|
||||
tests/test_spidermiddleware_offsite.py E501 E128 E111
|
||||
tests/test_spidermiddleware_output_chain.py E501 E226
|
||||
tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121
|
||||
tests/test_squeues.py E501 E701 E741
|
||||
tests/test_utils_asyncio.py E501
|
||||
tests/test_utils_conf.py E501 E303 E128
|
||||
tests/test_utils_curl.py E501
|
||||
tests/test_utils_datatypes.py E402 E501 E305
|
||||
tests/test_utils_defer.py E306 E261 E501 F841 E226
|
||||
tests/test_utils_defer.py E306 E501 F841 E226
|
||||
tests/test_utils_deprecate.py F841 E306 E501
|
||||
tests/test_utils_http.py E501 E502 E128 W504
|
||||
tests/test_utils_http.py E501 E128 W504
|
||||
tests/test_utils_iterators.py E501 E128 E129 E303 E241
|
||||
tests/test_utils_log.py E741 E226
|
||||
tests/test_utils_python.py E501 E303 E731 E701 E305
|
||||
|
|
@ -244,11 +241,11 @@ flake8-ignore =
|
|||
tests/test_utils_response.py E501
|
||||
tests/test_utils_signal.py E741 F841 E731 E226
|
||||
tests/test_utils_sitemap.py E128 E501 E124
|
||||
tests/test_utils_spider.py E261 E305
|
||||
tests/test_utils_spider.py E305
|
||||
tests/test_utils_template.py E305
|
||||
tests/test_utils_url.py E501 E127 E305 E211 E125 E501 E226 E241 E126 E123
|
||||
tests/test_webclient.py E501 E128 E122 E303 E402 E306 E226 E241 E123 E126
|
||||
tests/test_cmdline/__init__.py E502 E501
|
||||
tests/test_cmdline/__init__.py E501
|
||||
tests/test_settings/__init__.py E501 E128
|
||||
tests/test_spiderloader/__init__.py E128 E501
|
||||
tests/test_utils_misc/__init__.py E501
|
||||
|
|
|
|||
|
|
@ -1,16 +0,0 @@
|
|||
parsel>=1.5.0
|
||||
PyDispatcher>=2.0.5
|
||||
Twisted>=17.9.0
|
||||
w3lib>=1.17.0
|
||||
|
||||
pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'"
|
||||
queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'"
|
||||
cryptography>=2.0 # Earlier versions would fail to install
|
||||
|
||||
# Reference versions taken from
|
||||
# https://packages.ubuntu.com/xenial/python/
|
||||
# https://packages.ubuntu.com/xenial/zope/
|
||||
cssselect>=0.9.1
|
||||
lxml>=3.5.0
|
||||
service_identity>=16.0.0
|
||||
zope.interface>=4.1.3
|
||||
|
|
@ -67,7 +67,7 @@ def _pop_command_name(argv):
|
|||
|
||||
def _print_header(settings, inproject):
|
||||
if inproject:
|
||||
print("Scrapy %s - project: %s\n" % (scrapy.__version__, \
|
||||
print("Scrapy %s - project: %s\n" % (scrapy.__version__,
|
||||
settings['BOT_NAME']))
|
||||
else:
|
||||
print("Scrapy %s - no active project\n" % scrapy.__version__)
|
||||
|
|
@ -123,7 +123,7 @@ def execute(argv=None, settings=None):
|
|||
inproject = inside_project()
|
||||
cmds = _get_commands_dict(settings, inproject)
|
||||
cmdname = _pop_command_name(argv)
|
||||
parser = optparse.OptionParser(formatter=optparse.TitledHelpFormatter(), \
|
||||
parser = optparse.OptionParser(formatter=optparse.TitledHelpFormatter(),
|
||||
conflict_handler='resolve')
|
||||
if not cmdname:
|
||||
_print_commands(settings, inproject)
|
||||
|
|
|
|||
|
|
@ -24,12 +24,11 @@ class Command(ScrapyCommand):
|
|||
|
||||
def add_options(self, parser):
|
||||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("--spider", dest="spider",
|
||||
help="use this spider")
|
||||
parser.add_option("--headers", dest="headers", action="store_true", \
|
||||
help="print response HTTP headers instead of body")
|
||||
parser.add_option("--no-redirect", dest="no_redirect", action="store_true", \
|
||||
default=False, help="do not handle HTTP 3xx status codes and print response as-is")
|
||||
parser.add_option("--spider", dest="spider", help="use this spider")
|
||||
parser.add_option("--headers", dest="headers", action="store_true",
|
||||
help="print response HTTP headers instead of body")
|
||||
parser.add_option("--no-redirect", dest="no_redirect", action="store_true",
|
||||
default=False, help="do not handle HTTP 3xx status codes and print response as-is")
|
||||
|
||||
def _print_headers(self, headers, prefix):
|
||||
for key, values in headers.items():
|
||||
|
|
|
|||
|
|
@ -43,8 +43,8 @@ class Command(ScrapyCommand):
|
|||
return False
|
||||
|
||||
if not re.search(r'^[_a-zA-Z]\w*$', project_name):
|
||||
print('Error: Project names must begin with a letter and contain'\
|
||||
' only\nletters, numbers and underscores')
|
||||
print('Error: Project names must begin with a letter and contain'
|
||||
' only\nletters, numbers and underscores')
|
||||
elif _module_exists(project_name):
|
||||
print('Error: Module %r already exists' % project_name)
|
||||
else:
|
||||
|
|
|
|||
|
|
@ -86,8 +86,8 @@ class ReturnsContract(Contract):
|
|||
else:
|
||||
expected = '%s..%s' % (self.min_bound, self.max_bound)
|
||||
|
||||
raise ContractFail("Returned %s %s, expected %s" % \
|
||||
(occurrences, self.obj_name, expected))
|
||||
raise ContractFail("Returned %s %s, expected %s" %
|
||||
(occurrences, self.obj_name, expected))
|
||||
|
||||
|
||||
class ScrapesContract(Contract):
|
||||
|
|
|
|||
|
|
@ -16,6 +16,7 @@ from twisted.web.http import _DataLoss, PotentialDataLoss
|
|||
from twisted.web.client import Agent, ResponseDone, HTTPConnectionPool, ResponseFailed, URI
|
||||
from twisted.internet.endpoints import TCP4ClientEndpoint
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import Headers
|
||||
from scrapy.responsetypes import responsetypes
|
||||
from scrapy.core.downloader.webclient import _parse
|
||||
|
|
@ -285,6 +286,12 @@ class ScrapyAgent(object):
|
|||
scheme = _parse(request.url)[0]
|
||||
proxyHost = to_unicode(proxyHost)
|
||||
omitConnectTunnel = b'noconnect' in proxyParams
|
||||
if omitConnectTunnel:
|
||||
warnings.warn("Using HTTPS proxies in the noconnect mode is deprecated. "
|
||||
"If you use Crawlera, it doesn't require this mode anymore, "
|
||||
"so you should update scrapy-crawlera to 1.3.0+ "
|
||||
"and remove '?noconnect' from the Crawlera URL.",
|
||||
ScrapyDeprecationWarning)
|
||||
if scheme == b'https' and not omitConnectTunnel:
|
||||
proxyAuth = request.headers.get(b'Proxy-Authorization', None)
|
||||
proxyConf = (proxyHost, proxyPort, proxyAuth)
|
||||
|
|
|
|||
|
|
@ -32,8 +32,8 @@ def _get_boto_connection():
|
|||
|
||||
class S3DownloadHandler(object):
|
||||
|
||||
def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None, \
|
||||
httpdownloadhandler=HTTPDownloadHandler, **kw):
|
||||
def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None,
|
||||
httpdownloadhandler=HTTPDownloadHandler, **kw):
|
||||
|
||||
if not aws_access_key_id:
|
||||
aws_access_key_id = settings['AWS_ACCESS_KEY_ID']
|
||||
|
|
|
|||
|
|
@ -42,7 +42,7 @@ class ScrapyHTTPPageGetter(HTTPClient):
|
|||
delimiter = b'\n'
|
||||
|
||||
def connectionMade(self):
|
||||
self.headers = Headers() # bucket for response headers
|
||||
self.headers = Headers() # bucket for response headers
|
||||
|
||||
# Method command
|
||||
self.sendCommand(self.factory.method, self.factory.path)
|
||||
|
|
@ -88,9 +88,9 @@ class ScrapyHTTPPageGetter(HTTPClient):
|
|||
if self.factory.url.startswith(b'https'):
|
||||
self.transport.stopProducing()
|
||||
|
||||
self.factory.noPage(\
|
||||
defer.TimeoutError("Getting %s took longer than %s seconds." % \
|
||||
(self.factory.url, self.factory.timeout)))
|
||||
self.factory.noPage(
|
||||
defer.TimeoutError("Getting %s took longer than %s seconds." %
|
||||
(self.factory.url, self.factory.timeout)))
|
||||
|
||||
|
||||
class ScrapyHTTPClientFactory(HTTPClientFactory):
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ class Slot(object):
|
|||
|
||||
def __init__(self, start_requests, close_if_idle, nextcall, scheduler):
|
||||
self.closing = False
|
||||
self.inprogress = set() # requests in progress
|
||||
self.inprogress = set() # requests in progress
|
||||
self.start_requests = iter(start_requests)
|
||||
self.close_if_idle = close_if_idle
|
||||
self.nextcall = nextcall
|
||||
|
|
|
|||
|
|
@ -123,7 +123,7 @@ class Scraper(object):
|
|||
callback/errback"""
|
||||
assert isinstance(response, (Response, Failure))
|
||||
|
||||
dfd = self._scrape2(response, request, spider) # returns spiders processed output
|
||||
dfd = self._scrape2(response, request, spider) # returns spiders processed output
|
||||
dfd.addErrback(self.handle_spider_error, request, response, spider)
|
||||
dfd.addCallback(self.handle_spider_output, request, response, spider)
|
||||
return dfd
|
||||
|
|
|
|||
|
|
@ -44,7 +44,7 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
try:
|
||||
result = method(response=response, spider=spider)
|
||||
if result is not None:
|
||||
raise _InvalidOutput('Middleware {} must return None or raise an exception, got {}' \
|
||||
raise _InvalidOutput('Middleware {} must return None or raise an exception, got {}'
|
||||
.format(fname(method), type(result)))
|
||||
except _InvalidOutput:
|
||||
raise
|
||||
|
|
@ -69,7 +69,7 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
elif result is None:
|
||||
continue
|
||||
else:
|
||||
raise _InvalidOutput('Middleware {} must return None or an iterable, got {}' \
|
||||
raise _InvalidOutput('Middleware {} must return None or an iterable, got {}'
|
||||
.format(fname(method), type(result)))
|
||||
return _failure
|
||||
|
||||
|
|
@ -103,7 +103,7 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
if _isiterable(result):
|
||||
result = evaluate_iterable(result, method_index)
|
||||
else:
|
||||
raise _InvalidOutput('Middleware {} must return an iterable, got {}' \
|
||||
raise _InvalidOutput('Middleware {} must return an iterable, got {}'
|
||||
.format(fname(method), type(result)))
|
||||
|
||||
return chain(result, recovered)
|
||||
|
|
|
|||
|
|
@ -37,8 +37,9 @@ class HttpCompressionMiddleware(object):
|
|||
if content_encoding:
|
||||
encoding = content_encoding.pop()
|
||||
decoded_body = self._decode(response.body, encoding.lower())
|
||||
respcls = responsetypes.from_args(headers=response.headers, \
|
||||
url=response.url, body=decoded_body)
|
||||
respcls = responsetypes.from_args(
|
||||
headers=response.headers, url=response.url, body=decoded_body
|
||||
)
|
||||
kwargs = dict(cls=respcls, body=decoded_body)
|
||||
if issubclass(respcls, TextResponse):
|
||||
# force recalculating the encoding until we make sure the
|
||||
|
|
|
|||
|
|
@ -199,7 +199,7 @@ class CsvItemExporter(BaseItemExporter):
|
|||
line_buffering=False,
|
||||
write_through=True,
|
||||
encoding=self.encoding,
|
||||
newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034
|
||||
newline='' # Windows needs this https://github.com/scrapy/scrapy/issues/3034
|
||||
)
|
||||
self.csv_writer = csv.writer(self.stream, **kwargs)
|
||||
self._headers_not_written = True
|
||||
|
|
|
|||
|
|
@ -54,9 +54,9 @@ class CloseSpider(object):
|
|||
self.crawler.engine.close_spider(spider, 'closespider_pagecount')
|
||||
|
||||
def spider_opened(self, spider):
|
||||
self.task = reactor.callLater(self.close_on['timeout'], \
|
||||
self.crawler.engine.close_spider, spider, \
|
||||
reason='closespider_timeout')
|
||||
self.task = reactor.callLater(self.close_on['timeout'],
|
||||
self.crawler.engine.close_spider, spider,
|
||||
reason='closespider_timeout')
|
||||
|
||||
def item_scraped(self, item, spider):
|
||||
self.counter['itemcount'] += 1
|
||||
|
|
|
|||
|
|
@ -113,8 +113,8 @@ class Response(object_ref):
|
|||
It accepts the same arguments as ``Request.__init__`` method,
|
||||
but ``url`` can be a relative URL or a ``scrapy.link.Link`` object,
|
||||
not only an absolute URL.
|
||||
|
||||
:class:`~.TextResponse` provides a :meth:`~.TextResponse.follow`
|
||||
|
||||
:class:`~.TextResponse` provides a :meth:`~.TextResponse.follow`
|
||||
method which supports selectors in addition to absolute/relative URLs
|
||||
and Link objects.
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -125,7 +125,7 @@ class TextResponse(Response):
|
|||
Return a :class:`~.Request` instance to follow a link ``url``.
|
||||
It accepts the same arguments as ``Request.__init__`` method,
|
||||
but ``url`` can be not only an absolute URL, but also
|
||||
|
||||
|
||||
* a relative URL;
|
||||
* a scrapy.link.Link object (e.g. a link extractor result);
|
||||
* an attribute Selector (not SelectorList) - e.g.
|
||||
|
|
@ -133,7 +133,7 @@ class TextResponse(Response):
|
|||
``response.xpath('//img/@src')[0]``.
|
||||
* a Selector for ``<a>`` or ``<link>`` element, e.g.
|
||||
``response.css('a.my_link')[0]``.
|
||||
|
||||
|
||||
See :ref:`response-follow-example` for usage examples.
|
||||
"""
|
||||
if isinstance(url, parsel.Selector):
|
||||
|
|
|
|||
|
|
@ -7,10 +7,12 @@ For more info see docs/topics/link-extractors.rst
|
|||
"""
|
||||
import re
|
||||
from urllib.parse import urlparse
|
||||
from warnings import warn
|
||||
|
||||
from parsel.csstranslator import HTMLTranslator
|
||||
from w3lib.url import canonicalize_url
|
||||
|
||||
from scrapy.utils.deprecate import ScrapyDeprecationWarning
|
||||
from scrapy.utils.misc import arg_to_iter
|
||||
from scrapy.utils.url import (
|
||||
url_is_from_any_domain, url_has_any_extension,
|
||||
|
|
@ -44,14 +46,22 @@ IGNORED_EXTENSIONS = [
|
|||
|
||||
_re_type = type(re.compile("", 0))
|
||||
_matches = lambda url, regexs: any(r.search(url) for r in regexs)
|
||||
_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', \
|
||||
'file', 'ftp'}
|
||||
_is_valid_url = lambda url: url.split('://', 1)[0] in {'http', 'https', 'file', 'ftp'}
|
||||
|
||||
|
||||
class FilteringLinkExtractor(object):
|
||||
|
||||
_csstranslator = HTMLTranslator()
|
||||
|
||||
def __new__(cls, *args, **kwargs):
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
|
||||
if (issubclass(cls, FilteringLinkExtractor) and
|
||||
not issubclass(cls, LxmlLinkExtractor)):
|
||||
warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, '
|
||||
'please use scrapy.linkextractors.LinkExtractor instead',
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
return super(FilteringLinkExtractor, cls).__new__(cls)
|
||||
|
||||
def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains,
|
||||
restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text):
|
||||
|
||||
|
|
|
|||
|
|
@ -116,6 +116,14 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
|
|||
restrict_text=restrict_text)
|
||||
|
||||
def extract_links(self, response):
|
||||
"""Returns a list of :class:`~scrapy.link.Link` objects from the
|
||||
specified :class:`response <scrapy.http.Response>`.
|
||||
|
||||
Only links that match the settings passed to the ``__init__`` method of
|
||||
the link extractor are returned.
|
||||
|
||||
Duplicate links are omitted.
|
||||
"""
|
||||
base_url = get_base_url(response)
|
||||
if self.restrict_xpaths:
|
||||
docs = [subdoc
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ ERRORMSG = u"'Error processing %(item)s'"
|
|||
|
||||
class LogFormatter(object):
|
||||
"""Class for generating log messages for different actions.
|
||||
|
||||
|
||||
All methods must return a dictionary listing the parameters ``level``, ``msg``
|
||||
and ``args`` which are going to be used for constructing the log message when
|
||||
calling ``logging.log``.
|
||||
|
|
@ -48,7 +48,7 @@ class LogFormatter(object):
|
|||
}
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
def crawled(self, request, response, spider):
|
||||
"""Logs a message when the crawler finds a webpage."""
|
||||
request_flags = ' %s' % str(request.flags) if request.flags else ''
|
||||
|
|
|
|||
|
|
@ -73,8 +73,7 @@ class MailSender(object):
|
|||
part = MIMEBase(*mimetype.split('/'))
|
||||
part.set_payload(f.read())
|
||||
Encoders.encode_base64(part)
|
||||
part.add_header('Content-Disposition', 'attachment; filename="%s"' \
|
||||
% attach_name)
|
||||
part.add_header('Content-Disposition', 'attachment', filename=attach_name)
|
||||
msg.attach(part)
|
||||
else:
|
||||
msg.set_payload(body)
|
||||
|
|
|
|||
|
|
@ -65,8 +65,8 @@ class MiddlewareManager(object):
|
|||
return process_chain(self.methods[methodname], obj, *args)
|
||||
|
||||
def _process_chain_both(self, cb_methodname, eb_methodname, obj, *args):
|
||||
return process_chain_both(self.methods[cb_methodname], \
|
||||
self.methods[eb_methodname], obj, *args)
|
||||
return process_chain_both(self.methods[cb_methodname],
|
||||
self.methods[eb_methodname], obj, *args)
|
||||
|
||||
def open_spider(self, spider):
|
||||
return self._process_parallel('open_spider', spider)
|
||||
|
|
|
|||
|
|
@ -86,8 +86,8 @@ DOWNLOADER = 'scrapy.core.downloader.Downloader'
|
|||
DOWNLOADER_HTTPCLIENTFACTORY = 'scrapy.core.downloader.webclient.ScrapyHTTPClientFactory'
|
||||
DOWNLOADER_CLIENTCONTEXTFACTORY = 'scrapy.core.downloader.contextfactory.ScrapyClientContextFactory'
|
||||
DOWNLOADER_CLIENT_TLS_CIPHERS = 'DEFAULT'
|
||||
DOWNLOADER_CLIENT_TLS_METHOD = 'TLS' # Use highest TLS/SSL protocol version supported by the platform,
|
||||
# also allowing negotiation
|
||||
# Use highest TLS/SSL protocol version supported by the platform, also allowing negotiation:
|
||||
DOWNLOADER_CLIENT_TLS_METHOD = 'TLS'
|
||||
DOWNLOADER_CLIENT_TLS_VERBOSE_LOGGING = False
|
||||
|
||||
DOWNLOADER_MIDDLEWARES = {}
|
||||
|
|
|
|||
|
|
@ -16,7 +16,11 @@ from scrapy.utils.python import get_func_args
|
|||
from scrapy.utils.spider import iterate_spider_output
|
||||
|
||||
|
||||
def _identity(request, response):
|
||||
def _identity(x):
|
||||
return x
|
||||
|
||||
|
||||
def _identity_process_request(request, response):
|
||||
return request
|
||||
|
||||
|
||||
|
|
@ -32,17 +36,20 @@ _default_link_extractor = LinkExtractor()
|
|||
|
||||
class Rule(object):
|
||||
|
||||
def __init__(self, link_extractor=None, callback=None, cb_kwargs=None, follow=None, process_links=None, process_request=None):
|
||||
def __init__(self, link_extractor=None, callback=None, cb_kwargs=None, follow=None,
|
||||
process_links=None, process_request=None, errback=None):
|
||||
self.link_extractor = link_extractor or _default_link_extractor
|
||||
self.callback = callback
|
||||
self.errback = errback
|
||||
self.cb_kwargs = cb_kwargs or {}
|
||||
self.process_links = process_links
|
||||
self.process_request = process_request or _identity
|
||||
self.process_links = process_links or _identity
|
||||
self.process_request = process_request or _identity_process_request
|
||||
self.process_request_argcount = None
|
||||
self.follow = follow if follow is not None else not callback
|
||||
|
||||
def _compile(self, spider):
|
||||
self.callback = _get_method(self.callback, spider)
|
||||
self.errback = _get_method(self.errback, spider)
|
||||
self.process_links = _get_method(self.process_links, spider)
|
||||
self.process_request = _get_method(self.process_request, spider)
|
||||
self.process_request_argcount = len(get_func_args(self.process_request))
|
||||
|
|
@ -76,48 +83,59 @@ class CrawlSpider(Spider):
|
|||
def process_results(self, response, results):
|
||||
return results
|
||||
|
||||
def _build_request(self, rule, link):
|
||||
r = Request(url=link.url, callback=self._response_downloaded)
|
||||
r.meta.update(rule=rule, link_text=link.text)
|
||||
return r
|
||||
def _build_request(self, rule_index, link):
|
||||
return Request(
|
||||
url=link.url,
|
||||
callback=self._callback,
|
||||
errback=self._errback,
|
||||
meta=dict(rule=rule_index, link_text=link.text),
|
||||
)
|
||||
|
||||
def _requests_to_follow(self, response):
|
||||
if not isinstance(response, HtmlResponse):
|
||||
return
|
||||
seen = set()
|
||||
for n, rule in enumerate(self._rules):
|
||||
for rule_index, rule in enumerate(self._rules):
|
||||
links = [lnk for lnk in rule.link_extractor.extract_links(response)
|
||||
if lnk not in seen]
|
||||
if links and rule.process_links:
|
||||
links = rule.process_links(links)
|
||||
for link in links:
|
||||
for link in rule.process_links(links):
|
||||
seen.add(link)
|
||||
request = self._build_request(n, link)
|
||||
request = self._build_request(rule_index, link)
|
||||
yield rule._process_request(request, response)
|
||||
|
||||
def _response_downloaded(self, response):
|
||||
def _callback(self, response):
|
||||
rule = self._rules[response.meta['rule']]
|
||||
return self._parse_response(response, rule.callback, rule.cb_kwargs, rule.follow)
|
||||
|
||||
def _errback(self, failure):
|
||||
rule = self._rules[failure.request.meta['rule']]
|
||||
return self._handle_failure(failure, rule.errback)
|
||||
|
||||
def _parse_response(self, response, callback, cb_kwargs, follow=True):
|
||||
if callback:
|
||||
cb_res = callback(response, **cb_kwargs) or ()
|
||||
cb_res = self.process_results(response, cb_res)
|
||||
for requests_or_item in iterate_spider_output(cb_res):
|
||||
yield requests_or_item
|
||||
for request_or_item in iterate_spider_output(cb_res):
|
||||
yield request_or_item
|
||||
|
||||
if follow and self._follow_links:
|
||||
for request_or_item in self._requests_to_follow(response):
|
||||
yield request_or_item
|
||||
|
||||
def _handle_failure(self, failure, errback):
|
||||
if errback:
|
||||
results = errback(failure) or ()
|
||||
for request_or_item in iterate_spider_output(results):
|
||||
yield request_or_item
|
||||
|
||||
def _compile_rules(self):
|
||||
self._rules = [copy.copy(r) for r in self.rules]
|
||||
for rule in self._rules:
|
||||
rule._compile(self)
|
||||
self._rules = []
|
||||
for rule in self.rules:
|
||||
self._rules.append(copy.copy(rule))
|
||||
self._rules[-1]._compile(self)
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, *args, **kwargs):
|
||||
spider = super(CrawlSpider, cls).from_crawler(crawler, *args, **kwargs)
|
||||
spider._follow_links = crawler.settings.getbool(
|
||||
'CRAWLSPIDER_FOLLOW_LINKS', True)
|
||||
spider._follow_links = crawler.settings.getbool('CRAWLSPIDER_FOLLOW_LINKS', True)
|
||||
return spider
|
||||
|
|
|
|||
|
|
@ -100,8 +100,8 @@ class CSVFeedSpider(Spider):
|
|||
and the file's headers.
|
||||
"""
|
||||
|
||||
delimiter = None # When this is None, python's csv module's default delimiter is used
|
||||
quotechar = None # When this is None, python's csv module's default quotechar is used
|
||||
delimiter = None # When this is None, python's csv module's default delimiter is used
|
||||
quotechar = None # When this is None, python's csv module's default quotechar is used
|
||||
headers = None
|
||||
|
||||
def process_results(self, response, results):
|
||||
|
|
|
|||
|
|
@ -37,7 +37,7 @@ def build_component_list(compdict, custom=None, convert=update_classpath):
|
|||
"""Fail if a value in the components dict is not a real number or None."""
|
||||
for name, value in compdict.items():
|
||||
if value is not None and not isinstance(value, numbers.Real):
|
||||
raise ValueError('Invalid value {} for component {}, please provide ' \
|
||||
raise ValueError('Invalid value {} for component {}, please provide '
|
||||
'a real number or None instead'.format(value, name))
|
||||
|
||||
# BEGIN Backward compatibility for old (base, custom) call signature
|
||||
|
|
|
|||
|
|
@ -47,7 +47,7 @@ def _embed_ptpython_shell(namespace={}, banner=''):
|
|||
def _embed_standard_shell(namespace={}, banner=''):
|
||||
"""Start a standard python shell"""
|
||||
import code
|
||||
try: # readline module is only available on unix systems
|
||||
try: # readline module is only available on unix systems
|
||||
import readline
|
||||
except ImportError:
|
||||
pass
|
||||
|
|
@ -72,9 +72,9 @@ def get_shell_embed_func(shells=None, known_shells=None):
|
|||
"""Return the first acceptable shell-embed function
|
||||
from a given list of shell names.
|
||||
"""
|
||||
if shells is None: # list, preference order of shells
|
||||
if shells is None: # list, preference order of shells
|
||||
shells = DEFAULT_PYTHON_SHELLS.keys()
|
||||
if known_shells is None: # available embeddable shells
|
||||
if known_shells is None: # available embeddable shells
|
||||
known_shells = DEFAULT_PYTHON_SHELLS.copy()
|
||||
for shell in shells:
|
||||
if shell in known_shells:
|
||||
|
|
@ -97,5 +97,5 @@ def start_python_console(namespace=None, banner='', shells=None):
|
|||
shell = get_shell_embed_func(shells)
|
||||
if shell is not None:
|
||||
shell(namespace=namespace, banner=banner)
|
||||
except SystemExit: # raised when using exit() in python code.interact
|
||||
except SystemExit: # raised when using exit() in python code.interact
|
||||
pass
|
||||
|
|
|
|||
|
|
@ -11,4 +11,4 @@ from w3lib.html import * # noqa: F401
|
|||
|
||||
warnings.warn("Module `scrapy.utils.markup` is deprecated. "
|
||||
"Please import from `w3lib.html` instead.",
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
|
|
|
|||
|
|
@ -12,4 +12,4 @@ from w3lib.form import * # noqa: F401
|
|||
warnings.warn("Module `scrapy.utils.multipart` is deprecated. "
|
||||
"If you're using `encode_multipart` function, please use "
|
||||
"`urllib3.filepost.encode_multipart_formdata` instead",
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
|
|
|
|||
|
|
@ -254,8 +254,8 @@ ItemForm
|
|||
|
||||
#!python
|
||||
class MySiteForm(ItemForm):
|
||||
witdth = adaptor(ItemForm.witdh, default_unit='cm')
|
||||
volume = adaptor(ItemForm.witdh, default_unit='lt')
|
||||
width = adaptor(ItemForm.width, default_unit='cm')
|
||||
volume = adaptor(ItemForm.width, default_unit='lt')
|
||||
|
||||
ia['width'] = x.x('//p[@class="width"]')
|
||||
ia['volume'] = x.x('//p[@class="volume"]')
|
||||
|
|
|
|||
|
|
@ -53,7 +53,7 @@ Here's a simple proof-of-concept code of such script:
|
|||
# ... do something more interesting with scraped_items ...
|
||||
|
||||
The behaviour of the Scrapy crawler would be controller by the Scrapy settings,
|
||||
naturally, just like any typical scrapy project. But the default settings
|
||||
naturally, just like any typical Scrapy project. But the default settings
|
||||
should be sufficient so as to not require adding any specific setting. But, at
|
||||
the same time, you could do it if you need to, say, for specifying a custom
|
||||
middleware.
|
||||
|
|
|
|||
|
|
@ -185,7 +185,7 @@ These ideas translate to the following changes on the ``SpiderManager`` class:
|
|||
will return a spider class, not an instance. It's basically a ``__get__``
|
||||
to ``self._spiders``.
|
||||
|
||||
- All remaining functions should be deprecated or remove accordantly, since a
|
||||
- All remaining functions should be deprecated or remove accordingly, since a
|
||||
crawler reference is no longer needed.
|
||||
|
||||
- New helper ``get_spider_manager_class_from_scrapycfg`` in
|
||||
|
|
|
|||
|
|
@ -165,6 +165,12 @@ class Drop(Partial):
|
|||
request.finish()
|
||||
|
||||
|
||||
class ArbitraryLengthPayloadResource(LeafResource):
|
||||
|
||||
def render(self, request):
|
||||
return request.content.read()
|
||||
|
||||
|
||||
class Root(Resource):
|
||||
|
||||
def __init__(self):
|
||||
|
|
@ -178,6 +184,7 @@ class Root(Resource):
|
|||
self.putChild(b"echo", Echo())
|
||||
self.putChild(b"payload", PayloadResource())
|
||||
self.putChild(b"xpayload", EncodingResourceWrapper(PayloadResource(), [GzipEncoderFactory()]))
|
||||
self.putChild(b"alpayload", ArbitraryLengthPayloadResource())
|
||||
try:
|
||||
from tests import tests_datadir
|
||||
self.putChild(b"files", File(os.path.join(tests_datadir, 'test_site/files/')))
|
||||
|
|
|
|||
|
|
@ -1,14 +1,14 @@
|
|||
"""
|
||||
Some spiders used for testing and benchmarking
|
||||
"""
|
||||
|
||||
import time
|
||||
from urllib.parse import urlencode
|
||||
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.http import Request
|
||||
from scrapy.item import Item
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
from scrapy.spiders import Spider
|
||||
from scrapy.spiders.crawl import CrawlSpider, Rule
|
||||
|
||||
|
||||
class MockServerSpider(Spider):
|
||||
|
|
@ -184,3 +184,35 @@ class DuplicateStartRequestsSpider(MockServerSpider):
|
|||
|
||||
def parse(self, response):
|
||||
self.visited += 1
|
||||
|
||||
|
||||
class CrawlSpiderWithErrback(MockServerSpider, CrawlSpider):
|
||||
name = 'crawl_spider_with_errback'
|
||||
custom_settings = {
|
||||
'RETRY_HTTP_CODES': [], # no need to retry
|
||||
}
|
||||
rules = (
|
||||
Rule(LinkExtractor(), callback='callback', errback='errback', follow=True),
|
||||
)
|
||||
|
||||
def start_requests(self):
|
||||
test_body = b"""
|
||||
<html>
|
||||
<head><title>Page title<title></head>
|
||||
<body>
|
||||
<p><a href="/status?n=200">Item 200</a></p> <!-- callback -->
|
||||
<p><a href="/status?n=201">Item 201</a></p> <!-- callback -->
|
||||
<p><a href="/status?n=404">Item 404</a></p> <!-- errback -->
|
||||
<p><a href="/status?n=500">Item 500</a></p> <!-- errback -->
|
||||
<p><a href="/status?n=501">Item 501</a></p> <!-- errback -->
|
||||
</body>
|
||||
</html>
|
||||
"""
|
||||
url = self.mockserver.url("/alpayload")
|
||||
yield Request(url, method="POST", body=test_body)
|
||||
|
||||
def callback(self, response):
|
||||
self.logger.info('[callback] status %i', response.status)
|
||||
|
||||
def errback(self, failure):
|
||||
self.logger.info('[errback] status %i', failure.value.response.status)
|
||||
|
|
|
|||
|
|
@ -25,17 +25,15 @@ class CmdlineTest(unittest.TestCase):
|
|||
return comm.decode(encoding)
|
||||
|
||||
def test_default_settings(self):
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1'), \
|
||||
'default')
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1'), 'default')
|
||||
|
||||
def test_override_settings_using_set_arg(self):
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1', '-s', 'TEST1=override'), \
|
||||
'override')
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1', '-s',
|
||||
'TEST1=override'), 'override')
|
||||
|
||||
def test_override_settings_using_envvar(self):
|
||||
self.env['SCRAPY_TEST1'] = 'override'
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1'), \
|
||||
'override')
|
||||
self.assertEqual(self._execute('settings', '--get', 'TEST1'), 'override')
|
||||
|
||||
def test_profiling(self):
|
||||
path = tempfile.mkdtemp()
|
||||
|
|
|
|||
|
|
@ -29,6 +29,6 @@ class FetchTest(ProcessTest, SiteTest, unittest.TestCase):
|
|||
@defer.inlineCallbacks
|
||||
def test_headers(self):
|
||||
_, out, _ = yield self.execute([self.url('/text'), '--headers'])
|
||||
out = out.replace(b'\r', b'') # required on win32
|
||||
out = out.replace(b'\r', b'') # required on win32
|
||||
assert b'Server: TwistedWeb' in out, out
|
||||
assert b'Content-Type: text/plain' in out
|
||||
|
|
|
|||
|
|
@ -252,7 +252,7 @@ class ContractsManagerTest(unittest.TestCase):
|
|||
self.assertEqual(len(contracts), 3)
|
||||
self.assertEqual(frozenset(type(x) for x in contracts),
|
||||
frozenset([UrlContract, CallbackKeywordArgumentsContract, ReturnsContract]))
|
||||
|
||||
|
||||
contracts = self.conman.extract_contracts(spider.returns_item_cb_kwargs)
|
||||
self.assertEqual(len(contracts), 3)
|
||||
self.assertEqual(frozenset(type(x) for x in contracts),
|
||||
|
|
|
|||
|
|
@ -5,12 +5,12 @@ from testfixtures import LogCapture
|
|||
from twisted.internet import defer
|
||||
from twisted.trial.unittest import TestCase
|
||||
|
||||
from scrapy.http import Request
|
||||
from scrapy.crawler import CrawlerRunner
|
||||
from scrapy.http import Request
|
||||
from scrapy.utils.python import to_unicode
|
||||
from tests.spiders import FollowAllSpider, DelaySpider, SimpleSpider, \
|
||||
BrokenStartRequestsSpider, SingleRequestSpider, DuplicateStartRequestsSpider
|
||||
from tests.mockserver import MockServer
|
||||
from tests.spiders import (FollowAllSpider, DelaySpider, SimpleSpider, BrokenStartRequestsSpider,
|
||||
SingleRequestSpider, DuplicateStartRequestsSpider, CrawlSpiderWithErrback)
|
||||
|
||||
|
||||
class CrawlTestCase(TestCase):
|
||||
|
|
@ -30,25 +30,44 @@ class CrawlTestCase(TestCase):
|
|||
self.assertEqual(len(crawler.spider.urls_visited), 11) # 10 + start_url
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_delay(self):
|
||||
# short to long delays
|
||||
yield self._test_delay(0.2, False)
|
||||
yield self._test_delay(1, False)
|
||||
# randoms
|
||||
yield self._test_delay(0.2, True)
|
||||
yield self._test_delay(1, True)
|
||||
def test_fixed_delay(self):
|
||||
yield self._test_delay(total=3, delay=0.1)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def _test_delay(self, delay, randomize):
|
||||
settings = {"DOWNLOAD_DELAY": delay, 'RANDOMIZE_DOWNLOAD_DELAY': randomize}
|
||||
def test_randomized_delay(self):
|
||||
yield self._test_delay(total=3, delay=0.1, randomize=True)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def _test_delay(self, total, delay, randomize=False):
|
||||
crawl_kwargs = dict(
|
||||
maxlatency=delay * 2,
|
||||
mockserver=self.mockserver,
|
||||
total=total,
|
||||
)
|
||||
tolerance = (1 - (0.6 if randomize else 0.2))
|
||||
|
||||
settings = {"DOWNLOAD_DELAY": delay,
|
||||
'RANDOMIZE_DOWNLOAD_DELAY': randomize}
|
||||
crawler = CrawlerRunner(settings).create_crawler(FollowAllSpider)
|
||||
yield crawler.crawl(maxlatency=delay * 2, mockserver=self.mockserver)
|
||||
t = crawler.spider.times
|
||||
totaltime = t[-1] - t[0]
|
||||
avgd = totaltime / (len(t) - 1)
|
||||
tolerance = 0.6 if randomize else 0.2
|
||||
self.assertTrue(avgd > delay * (1 - tolerance),
|
||||
"download delay too small: %s" % avgd)
|
||||
yield crawler.crawl(**crawl_kwargs)
|
||||
times = crawler.spider.times
|
||||
total_time = times[-1] - times[0]
|
||||
average = total_time / (len(times) - 1)
|
||||
self.assertTrue(average > delay * tolerance,
|
||||
"download delay too small: %s" % average)
|
||||
|
||||
# Ensure that the same test parameters would cause a failure if no
|
||||
# download delay is set. Otherwise, it means we are using a combination
|
||||
# of ``total`` and ``delay`` values that are too small for the test
|
||||
# code above to have any meaning.
|
||||
settings["DOWNLOAD_DELAY"] = 0
|
||||
crawler = CrawlerRunner(settings).create_crawler(FollowAllSpider)
|
||||
yield crawler.crawl(**crawl_kwargs)
|
||||
times = crawler.spider.times
|
||||
total_time = times[-1] - times[0]
|
||||
average = total_time / (len(times) - 1)
|
||||
self.assertFalse(average > delay / tolerance,
|
||||
"test total or delay values are too small")
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_timeout_success(self):
|
||||
|
|
@ -277,3 +296,15 @@ with multiples lines
|
|||
|
||||
self._assert_retried(log)
|
||||
self.assertIn("Got response 200", str(log))
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawlspider_with_errback(self):
|
||||
self.runner.crawl(CrawlSpiderWithErrback, mockserver=self.mockserver)
|
||||
|
||||
with LogCapture() as log:
|
||||
yield self.runner.join()
|
||||
|
||||
self.assertIn("[callback] status 200", str(log))
|
||||
self.assertIn("[callback] status 201", str(log))
|
||||
self.assertIn("[errback] status 404", str(log))
|
||||
self.assertIn("[errback] status 500", str(log))
|
||||
|
|
|
|||
|
|
@ -33,7 +33,7 @@ from scrapy.responsetypes import responsetypes
|
|||
from scrapy.settings import Settings
|
||||
from scrapy.utils.test import get_crawler, skip_if_no_boto
|
||||
from scrapy.utils.python import to_bytes
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning
|
||||
|
||||
from tests.mockserver import MockServer, ssl_context_factory, Echo
|
||||
from tests.spiders import SingleRequestSpider
|
||||
|
|
@ -349,7 +349,7 @@ class HttpTestCase(unittest.TestCase):
|
|||
return self.download_request(request, Spider('foo')).addCallback(_test)
|
||||
|
||||
def test_payload(self):
|
||||
body = b'1'*100 # PayloadResource requires body length to be 100
|
||||
body = b'1'*100 # PayloadResource requires body length to be 100
|
||||
request = Request(self.getURL('payload'), method='POST', body=body)
|
||||
d = self.download_request(request, Spider('foo'))
|
||||
d.addCallback(lambda r: r.body)
|
||||
|
|
@ -695,7 +695,9 @@ class HttpProxyTestCase(unittest.TestCase):
|
|||
|
||||
http_proxy = '%s?noconnect' % self.getURL('')
|
||||
request = Request('https://example.com', meta={'proxy': http_proxy})
|
||||
return self.download_request(request, Spider('foo')).addCallback(_test)
|
||||
with self.assertWarnsRegex(ScrapyDeprecationWarning,
|
||||
r'Using HTTPS proxies in the noconnect mode is deprecated'):
|
||||
return self.download_request(request, Spider('foo')).addCallback(_test)
|
||||
|
||||
def test_download_without_proxy(self):
|
||||
def _test(response):
|
||||
|
|
@ -710,6 +712,9 @@ class HttpProxyTestCase(unittest.TestCase):
|
|||
class Http10ProxyTestCase(HttpProxyTestCase):
|
||||
download_handler_cls = HTTP10DownloadHandler
|
||||
|
||||
def test_download_with_proxy_https_noconnect(self):
|
||||
raise unittest.SkipTest('noconnect is not supported in HTTP10DownloadHandler')
|
||||
|
||||
|
||||
class Http11ProxyTestCase(HttpProxyTestCase):
|
||||
download_handler_cls = HTTP11DownloadHandler
|
||||
|
|
@ -800,8 +805,8 @@ class S3TestCase(unittest.TestCase):
|
|||
req = Request('s3://johnsmith/photos/puppy.jpg', headers={'Date': date})
|
||||
with self._mocked_date(date):
|
||||
httpreq = self.download_request(req, self.spider)
|
||||
self.assertEqual(httpreq.headers['Authorization'], \
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:xXjDGYUmKxnwqr5KXNPGldn5LbA=')
|
||||
self.assertEqual(httpreq.headers['Authorization'],
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:xXjDGYUmKxnwqr5KXNPGldn5LbA=')
|
||||
|
||||
def test_request_signing2(self):
|
||||
# puts an object into the johnsmith bucket.
|
||||
|
|
@ -813,21 +818,22 @@ class S3TestCase(unittest.TestCase):
|
|||
})
|
||||
with self._mocked_date(date):
|
||||
httpreq = self.download_request(req, self.spider)
|
||||
self.assertEqual(httpreq.headers['Authorization'], \
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:hcicpDDvL9SsO6AkvxqmIWkmOuQ=')
|
||||
self.assertEqual(httpreq.headers['Authorization'],
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:hcicpDDvL9SsO6AkvxqmIWkmOuQ=')
|
||||
|
||||
def test_request_signing3(self):
|
||||
# lists the content of the johnsmith bucket.
|
||||
date = 'Tue, 27 Mar 2007 19:42:41 +0000'
|
||||
req = Request('s3://johnsmith/?prefix=photos&max-keys=50&marker=puppy', \
|
||||
method='GET', headers={
|
||||
'User-Agent': 'Mozilla/5.0',
|
||||
'Date': date,
|
||||
})
|
||||
req = Request(
|
||||
's3://johnsmith/?prefix=photos&max-keys=50&marker=puppy',
|
||||
method='GET', headers={
|
||||
'User-Agent': 'Mozilla/5.0',
|
||||
'Date': date,
|
||||
})
|
||||
with self._mocked_date(date):
|
||||
httpreq = self.download_request(req, self.spider)
|
||||
self.assertEqual(httpreq.headers['Authorization'], \
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:jsRt/rhG+Vtp88HrYL706QhE4w4=')
|
||||
self.assertEqual(httpreq.headers['Authorization'],
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:jsRt/rhG+Vtp88HrYL706QhE4w4=')
|
||||
|
||||
def test_request_signing4(self):
|
||||
# fetches the access control policy sub-resource for the 'johnsmith' bucket.
|
||||
|
|
@ -836,8 +842,8 @@ class S3TestCase(unittest.TestCase):
|
|||
method='GET', headers={'Date': date})
|
||||
with self._mocked_date(date):
|
||||
httpreq = self.download_request(req, self.spider)
|
||||
self.assertEqual(httpreq.headers['Authorization'], \
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:thdUi9VAkzhkniLj96JIrOPGi0g=')
|
||||
self.assertEqual(httpreq.headers['Authorization'],
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:thdUi9VAkzhkniLj96JIrOPGi0g=')
|
||||
|
||||
def test_request_signing5(self):
|
||||
try:
|
||||
|
|
@ -850,11 +856,11 @@ class S3TestCase(unittest.TestCase):
|
|||
# deletes an object from the 'johnsmith' bucket using the
|
||||
# path-style and Date alternative.
|
||||
date = 'Tue, 27 Mar 2007 21:20:27 +0000'
|
||||
req = Request('s3://johnsmith/photos/puppy.jpg', \
|
||||
method='DELETE', headers={
|
||||
'Date': date,
|
||||
'x-amz-date': 'Tue, 27 Mar 2007 21:20:26 +0000',
|
||||
})
|
||||
req = Request(
|
||||
's3://johnsmith/photos/puppy.jpg', method='DELETE', headers={
|
||||
'Date': date,
|
||||
'x-amz-date': 'Tue, 27 Mar 2007 21:20:26 +0000',
|
||||
})
|
||||
with self._mocked_date(date):
|
||||
httpreq = self.download_request(req, self.spider)
|
||||
# botocore does not override Date with x-amz-date
|
||||
|
|
@ -864,25 +870,26 @@ class S3TestCase(unittest.TestCase):
|
|||
def test_request_signing6(self):
|
||||
# uploads an object to a CNAME style virtual hosted bucket with metadata.
|
||||
date = 'Tue, 27 Mar 2007 21:06:08 +0000'
|
||||
req = Request('s3://static.johnsmith.net:8080/db-backup.dat.gz', \
|
||||
method='PUT', headers={
|
||||
'User-Agent': 'curl/7.15.5',
|
||||
'Host': 'static.johnsmith.net:8080',
|
||||
'Date': date,
|
||||
'x-amz-acl': 'public-read',
|
||||
'content-type': 'application/x-download',
|
||||
'Content-MD5': '4gJE4saaMU4BqNR0kLY+lw==',
|
||||
'X-Amz-Meta-ReviewedBy': 'joe@johnsmith.net,jane@johnsmith.net',
|
||||
'X-Amz-Meta-FileChecksum': '0x02661779',
|
||||
'X-Amz-Meta-ChecksumAlgorithm': 'crc32',
|
||||
'Content-Disposition': 'attachment; filename=database.dat',
|
||||
'Content-Encoding': 'gzip',
|
||||
'Content-Length': '5913339',
|
||||
})
|
||||
req = Request(
|
||||
's3://static.johnsmith.net:8080/db-backup.dat.gz',
|
||||
method='PUT', headers={
|
||||
'User-Agent': 'curl/7.15.5',
|
||||
'Host': 'static.johnsmith.net:8080',
|
||||
'Date': date,
|
||||
'x-amz-acl': 'public-read',
|
||||
'content-type': 'application/x-download',
|
||||
'Content-MD5': '4gJE4saaMU4BqNR0kLY+lw==',
|
||||
'X-Amz-Meta-ReviewedBy': 'joe@johnsmith.net,jane@johnsmith.net',
|
||||
'X-Amz-Meta-FileChecksum': '0x02661779',
|
||||
'X-Amz-Meta-ChecksumAlgorithm': 'crc32',
|
||||
'Content-Disposition': 'attachment; filename=database.dat',
|
||||
'Content-Encoding': 'gzip',
|
||||
'Content-Length': '5913339',
|
||||
})
|
||||
with self._mocked_date(date):
|
||||
httpreq = self.download_request(req, self.spider)
|
||||
self.assertEqual(httpreq.headers['Authorization'], \
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:C0FlOtU8Ylb9KDTpZqYkZPX91iI=')
|
||||
self.assertEqual(httpreq.headers['Authorization'],
|
||||
b'AWS 0PN5J17HBGZHT7JJ3X82:C0FlOtU8Ylb9KDTpZqYkZPX91iI=')
|
||||
|
||||
def test_request_signing7(self):
|
||||
# ensure that spaces are quoted properly before signing
|
||||
|
|
|
|||
|
|
@ -124,7 +124,7 @@ class MaxRetryTimesTest(unittest.TestCase):
|
|||
|
||||
# SETTINGS: meta(max_retry_times) = 0
|
||||
meta_max_retry_times = 0
|
||||
|
||||
|
||||
req = Request(self.invalid_url, meta={'max_retry_times': meta_max_retry_times})
|
||||
self._test_retry(req, DNSLookupError('foo'), meta_max_retry_times)
|
||||
|
||||
|
|
@ -137,7 +137,7 @@ class MaxRetryTimesTest(unittest.TestCase):
|
|||
self._test_retry(req, DNSLookupError('foo'), self.mw.max_retry_times)
|
||||
|
||||
def test_with_metakey_greater(self):
|
||||
|
||||
|
||||
# SETINGS: RETRY_TIMES < meta(max_retry_times)
|
||||
self.mw.max_retry_times = 2
|
||||
meta_max_retry_times = 3
|
||||
|
|
@ -149,7 +149,7 @@ class MaxRetryTimesTest(unittest.TestCase):
|
|||
self._test_retry(req2, DNSLookupError('foo'), self.mw.max_retry_times)
|
||||
|
||||
def test_with_metakey_lesser(self):
|
||||
|
||||
|
||||
# SETINGS: RETRY_TIMES > meta(max_retry_times)
|
||||
self.mw.max_retry_times = 5
|
||||
meta_max_retry_times = 4
|
||||
|
|
@ -165,14 +165,14 @@ class MaxRetryTimesTest(unittest.TestCase):
|
|||
# SETTINGS: meta(max_retry_times) = 4
|
||||
meta_max_retry_times = 4
|
||||
|
||||
req = Request(self.invalid_url, meta= \
|
||||
{'max_retry_times': meta_max_retry_times, 'dont_retry': True})
|
||||
req = Request(self.invalid_url, meta={
|
||||
'max_retry_times': meta_max_retry_times, 'dont_retry': True
|
||||
})
|
||||
|
||||
self._test_retry(req, DNSLookupError('foo'), 0)
|
||||
|
||||
|
||||
def _test_retry(self, req, exception, max_retry_times):
|
||||
|
||||
|
||||
for i in range(0, max_retry_times):
|
||||
req = self.mw.process_exception(req, exception, self.spider)
|
||||
assert isinstance(req, Request)
|
||||
|
|
|
|||
|
|
@ -142,12 +142,12 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
|
||||
r1 = Request('http://scrapytest.org/index.html')
|
||||
r2 = Request('http://scrapytest.org/index.html')
|
||||
|
||||
|
||||
dupefilter.log(r1, spider)
|
||||
dupefilter.log(r2, spider)
|
||||
|
||||
assert crawler.stats.get_value('dupefilter/filtered') == 2
|
||||
l.check_present(('scrapy.dupefilters', 'DEBUG',
|
||||
l.check_present(('scrapy.dupefilters', 'DEBUG',
|
||||
('Filtered duplicate request: <GET http://scrapytest.org/index.html>'
|
||||
' - no more duplicates will be shown'
|
||||
' (see DUPEFILTER_DEBUG to show all duplicates)')))
|
||||
|
|
@ -169,7 +169,7 @@ class RFPDupeFilterTest(unittest.TestCase):
|
|||
r2 = Request('http://scrapytest.org/index.html',
|
||||
headers={'Referer': 'http://scrapytest.org/INDEX.html'}
|
||||
)
|
||||
|
||||
|
||||
dupefilter.log(r1, spider)
|
||||
dupefilter.log(r2, spider)
|
||||
|
||||
|
|
|
|||
|
|
@ -91,8 +91,8 @@ def start_test_site(debug=False):
|
|||
|
||||
port = reactor.listenTCP(0, server.Site(r), interface="127.0.0.1")
|
||||
if debug:
|
||||
print("Test server running at http://localhost:%d/ - hit Ctrl-C to finish." \
|
||||
% port.getHost().port)
|
||||
print("Test server running at http://localhost:%d/ - hit Ctrl-C to finish."
|
||||
% port.getHost().port)
|
||||
return port
|
||||
|
||||
|
||||
|
|
@ -179,7 +179,6 @@ class EngineTest(unittest.TestCase):
|
|||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler(self):
|
||||
|
||||
for spider in TestSpider, DictItemsSpider:
|
||||
self.run = CrawlerRun(spider)
|
||||
yield self.run.run()
|
||||
|
|
@ -189,11 +188,15 @@ class EngineTest(unittest.TestCase):
|
|||
self._assert_scraped_items()
|
||||
self._assert_signals_catched()
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler_dupefilter(self):
|
||||
self.run = CrawlerRun(TestDupeFilterSpider)
|
||||
yield self.run.run()
|
||||
self._assert_scheduled_requests(urls_to_visit=7)
|
||||
self._assert_dropped_requests()
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_crawler_itemerror(self):
|
||||
self.run = CrawlerRun(ItemZeroDivisionErrorSpider)
|
||||
yield self.run.run()
|
||||
self._assert_items_error()
|
||||
|
|
@ -271,7 +274,6 @@ class EngineTest(unittest.TestCase):
|
|||
self.run.signals_catched[signals.spider_opened])
|
||||
self.assertEqual({'spider': self.run.spider},
|
||||
self.run.signals_catched[signals.spider_idle])
|
||||
self.run.signals_catched[signals.spider_closed].pop('spider_stats', None) # XXX: remove for scrapy 0.17
|
||||
self.assertEqual({'spider': self.run.spider, 'reason': 'finished'},
|
||||
self.run.signals_catched[signals.spider_closed])
|
||||
|
||||
|
|
|
|||
|
|
@ -149,7 +149,7 @@ class RequestTest(unittest.TestCase):
|
|||
|
||||
r2 = self.request_class(url="http://www.example.com/", body=b"")
|
||||
assert isinstance(r2.body, bytes)
|
||||
self.assertEqual(r2.encoding, 'utf-8') # default encoding
|
||||
self.assertEqual(r2.encoding, 'utf-8') # default encoding
|
||||
|
||||
r3 = self.request_class(url="http://www.example.com/", body=u"Price: \xa3100", encoding='utf-8')
|
||||
assert isinstance(r3.body, bytes)
|
||||
|
|
@ -622,8 +622,9 @@ class FormRequestTest(RequestTest):
|
|||
<input type="hidden" name="two" value="3">
|
||||
<input type="submit" name="clickable2" value="clicked2">
|
||||
</form>""")
|
||||
req = self.request_class.from_response(response, formdata={'two': '2'}, \
|
||||
clickdata={'name': 'clickable2'})
|
||||
req = self.request_class.from_response(
|
||||
response, formdata={'two': '2'}, clickdata={'name': 'clickable2'}
|
||||
)
|
||||
fs = _qs(req)
|
||||
self.assertEqual(fs[b'clickable2'], [b'clicked2'])
|
||||
self.assertFalse(b'clickable1' in fs, fs)
|
||||
|
|
@ -671,8 +672,9 @@ class FormRequestTest(RequestTest):
|
|||
<input type="hidden" name="one" value="clicked1">
|
||||
<input type="hidden" name="two" value="clicked2">
|
||||
</form>""")
|
||||
req = self.request_class.from_response(response, \
|
||||
clickdata={u'name': u'clickable', u'value': u'clicked2'})
|
||||
req = self.request_class.from_response(
|
||||
response, clickdata={u'name': u'clickable', u'value': u'clicked2'}
|
||||
)
|
||||
fs = _qs(req)
|
||||
self.assertEqual(fs[b'clickable'], [b'clicked2'])
|
||||
self.assertEqual(fs[b'one'], [b'clicked1'])
|
||||
|
|
@ -686,8 +688,9 @@ class FormRequestTest(RequestTest):
|
|||
<input type="hidden" name="poundsign" value="\u00a3">
|
||||
<input type="hidden" name="eurosign" value="\u20ac">
|
||||
</form>""")
|
||||
req = self.request_class.from_response(response, \
|
||||
clickdata={u'name': u'price in \u00a3'})
|
||||
req = self.request_class.from_response(
|
||||
response, clickdata={u'name': u'price in \u00a3'}
|
||||
)
|
||||
fs = _qs(req, to_unicode=True)
|
||||
self.assertTrue(fs[u'price in \u00a3'])
|
||||
|
||||
|
|
@ -700,8 +703,9 @@ class FormRequestTest(RequestTest):
|
|||
<input type="hidden" name="yensign" value="\u00a5">
|
||||
</form>""",
|
||||
encoding='latin1')
|
||||
req = self.request_class.from_response(response, \
|
||||
clickdata={u'name': u'price in \u00a5'})
|
||||
req = self.request_class.from_response(
|
||||
response, clickdata={u'name': u'price in \u00a5'}
|
||||
)
|
||||
fs = _qs(req, to_unicode=True, encoding='latin1')
|
||||
self.assertTrue(fs[u'price in \u00a5'])
|
||||
|
||||
|
|
@ -716,8 +720,9 @@ class FormRequestTest(RequestTest):
|
|||
<input type="hidden" name="field2" value="value2">
|
||||
</form>
|
||||
""")
|
||||
req = self.request_class.from_response(response, formname='form2', \
|
||||
clickdata={u'name': u'clickable'})
|
||||
req = self.request_class.from_response(
|
||||
response, formname='form2', clickdata={u'name': u'clickable'}
|
||||
)
|
||||
fs = _qs(req)
|
||||
self.assertEqual(fs[b'clickable'], [b'clicked2'])
|
||||
self.assertEqual(fs[b'field2'], [b'value2'])
|
||||
|
|
@ -725,8 +730,9 @@ class FormRequestTest(RequestTest):
|
|||
|
||||
def test_from_response_override_clickable(self):
|
||||
response = _buildresponse('''<form><input type="submit" name="clickme" value="one"> </form>''')
|
||||
req = self.request_class.from_response(response, \
|
||||
formdata={'clickme': 'two'}, clickdata={'name': 'clickme'})
|
||||
req = self.request_class.from_response(
|
||||
response, formdata={'clickme': 'two'}, clickdata={'name': 'clickme'}
|
||||
)
|
||||
fs = _qs(req)
|
||||
self.assertEqual(fs[b'clickme'], [b'two'])
|
||||
|
||||
|
|
@ -853,7 +859,7 @@ class FormRequestTest(RequestTest):
|
|||
<form name="form2" action="post.php" method="POST">
|
||||
<input type="hidden" name="two" value="2">
|
||||
</form>""")
|
||||
self.assertRaises(IndexError, self.request_class.from_response, \
|
||||
self.assertRaises(IndexError, self.request_class.from_response,
|
||||
response, formname="form3", formnumber=2)
|
||||
|
||||
def test_from_response_formid_exists(self):
|
||||
|
|
@ -907,7 +913,7 @@ class FormRequestTest(RequestTest):
|
|||
<form id="form2" name="form2" action="post.php" method="POST">
|
||||
<input type="hidden" name="two" value="2">
|
||||
</form>""")
|
||||
self.assertRaises(IndexError, self.request_class.from_response, \
|
||||
self.assertRaises(IndexError, self.request_class.from_response,
|
||||
response, formid="form3", formnumber=2)
|
||||
|
||||
def test_from_response_select(self):
|
||||
|
|
|
|||
|
|
@ -316,8 +316,8 @@ class TextResponseTest(BaseResponseTest):
|
|||
assert u'SUFFIX' in r.text, repr(r.text)
|
||||
|
||||
# Do not destroy html tags due to encoding bugs
|
||||
r = self.response_class("http://example.com", encoding='utf-8', \
|
||||
body=b'\xf0<span>value</span>')
|
||||
r = self.response_class("http://example.com", encoding='utf-8',
|
||||
body=b'\xf0<span>value</span>')
|
||||
assert u'<span>value</span>' in r.text, repr(r.text)
|
||||
|
||||
# FIXME: This test should pass once we stop using BeautifulSoup's UnicodeDammit in TextResponse
|
||||
|
|
|
|||
|
|
@ -1,10 +1,13 @@
|
|||
import re
|
||||
import unittest
|
||||
from warnings import catch_warnings
|
||||
|
||||
import pytest
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import HtmlResponse, XmlResponse
|
||||
from scrapy.link import Link
|
||||
from scrapy.linkextractors import FilteringLinkExtractor
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
|
||||
from tests import get_testdata
|
||||
|
||||
|
|
@ -506,3 +509,32 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
|
|||
@pytest.mark.xfail
|
||||
def test_restrict_xpaths_with_html_entities(self):
|
||||
super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities()
|
||||
|
||||
def test_filteringlinkextractor_deprecation_warning(self):
|
||||
"""Make sure the FilteringLinkExtractor deprecation warning is not
|
||||
issued for LxmlLinkExtractor"""
|
||||
with catch_warnings(record=True) as warnings:
|
||||
LxmlLinkExtractor()
|
||||
self.assertEqual(len(warnings), 0)
|
||||
|
||||
class SubclassedLxmlLinkExtractor(LxmlLinkExtractor):
|
||||
pass
|
||||
|
||||
SubclassedLxmlLinkExtractor()
|
||||
self.assertEqual(len(warnings), 0)
|
||||
|
||||
|
||||
class FilteringLinkExtractorTest(unittest.TestCase):
|
||||
|
||||
def test_deprecation_warning(self):
|
||||
args = [None] * 10
|
||||
with catch_warnings(record=True) as warnings:
|
||||
FilteringLinkExtractor(*args)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
with catch_warnings(record=True) as warnings:
|
||||
class SubclassedFilteringLinkExtractor(FilteringLinkExtractor):
|
||||
pass
|
||||
SubclassedFilteringLinkExtractor(*args)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
|
|
|
|||
|
|
@ -58,7 +58,6 @@ class FilesPipelineTestCase(unittest.TestCase):
|
|||
self.assertEqual(file_path(Request("data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAR0AAACxCAMAAADOHZloAAACClBMVEX/\
|
||||
//+F0tzCwMK76ZKQ21AMqr7oAAC96JvD5aWM2kvZ78J0N7fmAAC46Y4Ap7y")),
|
||||
'full/178059cbeba2e34120a67f2dc1afc3ecc09b61cb.png')
|
||||
|
||||
|
||||
def test_fs_store(self):
|
||||
assert isinstance(self.pipeline.store, FSFilesStore)
|
||||
|
|
|
|||
|
|
@ -238,10 +238,10 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase):
|
|||
self.assertEqual(new_item['results'], [(True, rsp1), (False, fail)])
|
||||
m = self.pipe._mockcalled
|
||||
# only once
|
||||
self.assertEqual(m[0], 'get_media_requests') # first hook called
|
||||
self.assertEqual(m[0], 'get_media_requests') # first hook called
|
||||
self.assertEqual(m.count('get_media_requests'), 1)
|
||||
self.assertEqual(m.count('item_completed'), 1)
|
||||
self.assertEqual(m[-1], 'item_completed') # last hook called
|
||||
self.assertEqual(m[-1], 'item_completed') # last hook called
|
||||
# twice, one per request
|
||||
self.assertEqual(m.count('media_to_download'), 2)
|
||||
# one to handle success and other for failure
|
||||
|
|
@ -252,7 +252,7 @@ class MediaPipelineTestCase(BaseMediaPipelineTestCase):
|
|||
def test_get_media_requests(self):
|
||||
# returns single Request (without callback)
|
||||
req = Request('http://url')
|
||||
item = dict(requests=req) # pass a single item
|
||||
item = dict(requests=req) # pass a single item
|
||||
new_item = yield self.pipe.process_item(item, self.spider)
|
||||
assert new_item is item
|
||||
assert request_fingerprint(req) in self.info.downloaded
|
||||
|
|
|
|||
|
|
@ -0,0 +1,71 @@
|
|||
from twisted.internet import defer
|
||||
from twisted.internet.defer import Deferred
|
||||
from twisted.trial import unittest
|
||||
|
||||
from scrapy import Spider, signals, Request
|
||||
from scrapy.utils.test import get_crawler
|
||||
|
||||
from tests.mockserver import MockServer
|
||||
|
||||
|
||||
class SimplePipeline:
|
||||
def process_item(self, item, spider):
|
||||
item['pipeline_passed'] = True
|
||||
return item
|
||||
|
||||
|
||||
class DeferredPipeline:
|
||||
def cb(self, item):
|
||||
item['pipeline_passed'] = True
|
||||
return item
|
||||
|
||||
def process_item(self, item, spider):
|
||||
d = Deferred()
|
||||
d.addCallback(self.cb)
|
||||
d.callback(item)
|
||||
return d
|
||||
|
||||
|
||||
class ItemSpider(Spider):
|
||||
name = 'itemspider'
|
||||
|
||||
def start_requests(self):
|
||||
yield Request(self.mockserver.url('/status?n=200'))
|
||||
|
||||
def parse(self, response):
|
||||
return {'field': 42}
|
||||
|
||||
|
||||
class PipelineTestCase(unittest.TestCase):
|
||||
def setUp(self):
|
||||
self.mockserver = MockServer()
|
||||
self.mockserver.__enter__()
|
||||
|
||||
def tearDown(self):
|
||||
self.mockserver.__exit__(None, None, None)
|
||||
|
||||
def _on_item_scraped(self, item):
|
||||
self.assertIsInstance(item, dict)
|
||||
self.assertTrue(item.get('pipeline_passed'))
|
||||
self.items.append(item)
|
||||
|
||||
def _create_crawler(self, pipeline_class):
|
||||
settings = {
|
||||
'ITEM_PIPELINES': {__name__ + '.' + pipeline_class.__name__: 1},
|
||||
}
|
||||
crawler = get_crawler(ItemSpider, settings)
|
||||
crawler.signals.connect(self._on_item_scraped, signals.item_scraped)
|
||||
self.items = []
|
||||
return crawler
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_simple_pipeline(self):
|
||||
crawler = self._create_crawler(SimplePipeline)
|
||||
yield crawler.crawl(mockserver=self.mockserver)
|
||||
self.assertEqual(len(self.items), 1)
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_deferred_pipeline(self):
|
||||
crawler = self._create_crawler(DeferredPipeline)
|
||||
yield crawler.crawl(mockserver=self.mockserver)
|
||||
self.assertEqual(len(self.items), 1)
|
||||
|
|
@ -108,32 +108,6 @@ class ProxyConnectTestCase(TestCase):
|
|||
echo = json.loads(crawler.spider.meta['responses'][0].text)
|
||||
self.assertTrue('Proxy-Authorization' not in echo['headers'])
|
||||
|
||||
# The noconnect mode isn't supported by the current mitmproxy, it returns
|
||||
# "Invalid request scheme: https" as it doesn't seem to support full URLs in GET at all,
|
||||
# and it's not clear what behavior is intended by Scrapy and by mitmproxy here.
|
||||
# https://github.com/mitmproxy/mitmproxy/issues/848 may be related.
|
||||
# The Scrapy noconnect mode was required, at least in the past, to work with Crawlera,
|
||||
# and https://github.com/scrapy-plugins/scrapy-crawlera/pull/44 seems to be related.
|
||||
|
||||
@pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests')
|
||||
@defer.inlineCallbacks
|
||||
def test_https_noconnect(self):
|
||||
proxy = os.environ['https_proxy']
|
||||
os.environ['https_proxy'] = proxy + '?noconnect'
|
||||
crawler = get_crawler(SimpleSpider)
|
||||
with LogCapture() as l:
|
||||
yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
|
||||
self._assert_got_response_code(200, l)
|
||||
|
||||
@pytest.mark.xfail(reason='mitmproxy gives an error for noconnect requests')
|
||||
@defer.inlineCallbacks
|
||||
def test_https_noconnect_auth_error(self):
|
||||
os.environ['https_proxy'] = _wrong_credentials(os.environ['https_proxy']) + '?noconnect'
|
||||
crawler = get_crawler(SimpleSpider)
|
||||
with LogCapture() as l:
|
||||
yield crawler.crawl(self.mockserver.url("/status?n=200", is_secure=True))
|
||||
self._assert_got_response_code(407, l)
|
||||
|
||||
def _assert_got_response_code(self, code, log):
|
||||
print(log)
|
||||
self.assertEqual(str(log).count('Crawled (%d)' % code), 1)
|
||||
|
|
|
|||
|
|
@ -44,7 +44,7 @@ class BaseRobotParserTest:
|
|||
|
||||
def test_allowed_wildcards(self):
|
||||
robotstxt_robotstxt_body = """User-agent: first
|
||||
Disallow: /disallowed/*/end$
|
||||
Disallow: /disallowed/*/end$
|
||||
|
||||
User-agent: second
|
||||
Allow: /*allowed
|
||||
|
|
|
|||
|
|
@ -73,7 +73,7 @@ class TestOffsiteMiddleware4(TestOffsiteMiddleware3):
|
|||
|
||||
|
||||
class TestOffsiteMiddleware5(TestOffsiteMiddleware4):
|
||||
|
||||
|
||||
def test_get_host_regex(self):
|
||||
self.spider.allowed_domains = ['http://scrapytest.org', 'scrapy.org', 'scrapy.test.org']
|
||||
with warnings.catch_warnings(record=True) as w:
|
||||
|
|
|
|||
|
|
@ -156,7 +156,7 @@ class GeneratorFailMiddleware:
|
|||
r['processed'].append('{}.process_spider_output'.format(self.__class__.__name__))
|
||||
yield r
|
||||
raise LookupError()
|
||||
|
||||
|
||||
def process_spider_exception(self, response, exception, spider):
|
||||
method = '{}.process_spider_exception'.format(self.__class__.__name__)
|
||||
spider.logger.info('%s: %s caught', method, exception.__class__.__name__)
|
||||
|
|
@ -264,7 +264,7 @@ class TestSpiderMiddleware(TestCase):
|
|||
@classmethod
|
||||
def tearDownClass(cls):
|
||||
cls.mockserver.__exit__(None, None, None)
|
||||
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def crawl_log(self, spider):
|
||||
crawler = get_crawler(spider)
|
||||
|
|
@ -308,7 +308,7 @@ class TestSpiderMiddleware(TestCase):
|
|||
self.assertIn("{'from': 'errback'}", str(log1))
|
||||
self.assertNotIn("{'from': 'callback'}", str(log1))
|
||||
self.assertIn("'item_scraped_count': 1", str(log1))
|
||||
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_generator_callback(self):
|
||||
"""
|
||||
|
|
@ -319,7 +319,7 @@ class TestSpiderMiddleware(TestCase):
|
|||
log2 = yield self.crawl_log(GeneratorCallbackSpider)
|
||||
self.assertIn("Middleware: ImportError exception caught", str(log2))
|
||||
self.assertIn("'item_scraped_count': 2", str(log2))
|
||||
|
||||
|
||||
@defer.inlineCallbacks
|
||||
def test_not_a_generator_callback(self):
|
||||
"""
|
||||
|
|
|
|||
|
|
@ -548,8 +548,8 @@ class TestReferrerOnRedirect(TestRefererMiddleware):
|
|||
(301, 'http://scrapytest.org/3'),
|
||||
(301, 'http://scrapytest.org/4'),
|
||||
),
|
||||
b'http://scrapytest.org/1', # expected initial referer
|
||||
b'http://scrapytest.org/1', # expected referer for the redirection request
|
||||
b'http://scrapytest.org/1', # expected initial referer
|
||||
b'http://scrapytest.org/1', # expected referer for the redirection request
|
||||
),
|
||||
( 'https://scrapytest.org/1',
|
||||
'https://scrapytest.org/2',
|
||||
|
|
@ -609,8 +609,8 @@ class TestReferrerOnRedirectNoReferrer(TestReferrerOnRedirect):
|
|||
(301, 'http://scrapytest.org/3'),
|
||||
(301, 'http://scrapytest.org/4'),
|
||||
),
|
||||
None, # expected initial "Referer"
|
||||
None, # expected "Referer" for the redirection request
|
||||
None, # expected initial "Referer"
|
||||
None, # expected "Referer" for the redirection request
|
||||
),
|
||||
( 'https://scrapytest.org/1',
|
||||
'https://scrapytest.org/2',
|
||||
|
|
@ -648,8 +648,8 @@ class TestReferrerOnRedirectSameOrigin(TestReferrerOnRedirect):
|
|||
(301, 'http://scrapytest.org/103'),
|
||||
(301, 'http://scrapytest.org/104'),
|
||||
),
|
||||
b'http://scrapytest.org/101', # expected initial "Referer"
|
||||
b'http://scrapytest.org/101', # expected referer for the redirection request
|
||||
b'http://scrapytest.org/101', # expected initial "Referer"
|
||||
b'http://scrapytest.org/101', # expected referer for the redirection request
|
||||
),
|
||||
( 'https://scrapytest.org/201',
|
||||
'https://scrapytest.org/202',
|
||||
|
|
@ -757,8 +757,8 @@ class TestReferrerOnRedirectOriginWhenCrossOrigin(TestReferrerOnRedirect):
|
|||
(301, 'http://scrapytest.org/103'),
|
||||
(301, 'http://scrapytest.org/104'),
|
||||
),
|
||||
b'http://scrapytest.org/101', # expected initial referer
|
||||
b'http://scrapytest.org/101', # expected referer for the redirection request
|
||||
b'http://scrapytest.org/101', # expected initial referer
|
||||
b'http://scrapytest.org/101', # expected referer for the redirection request
|
||||
),
|
||||
( 'https://scrapytest.org/201',
|
||||
'https://scrapytest.org/202',
|
||||
|
|
@ -827,8 +827,8 @@ class TestReferrerOnRedirectStrictOriginWhenCrossOrigin(TestReferrerOnRedirect):
|
|||
(301, 'http://scrapytest.org/103'),
|
||||
(301, 'http://scrapytest.org/104'),
|
||||
),
|
||||
b'http://scrapytest.org/101', # expected initial referer
|
||||
b'http://scrapytest.org/101', # expected referer for the redirection request
|
||||
b'http://scrapytest.org/101', # expected initial referer
|
||||
b'http://scrapytest.org/101', # expected referer for the redirection request
|
||||
),
|
||||
( 'https://scrapytest.org/201',
|
||||
'https://scrapytest.org/202',
|
||||
|
|
|
|||
|
|
@ -14,8 +14,8 @@ class MustbeDeferredTest(unittest.TestCase):
|
|||
return steps
|
||||
|
||||
dfd = mustbe_deferred(_append, 1)
|
||||
dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred
|
||||
steps.append(2) # add another value, that should be catched by assertEqual
|
||||
dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred
|
||||
steps.append(2) # add another value, that should be catched by assertEqual
|
||||
return dfd
|
||||
|
||||
def test_unfired_deferred(self):
|
||||
|
|
@ -27,8 +27,8 @@ class MustbeDeferredTest(unittest.TestCase):
|
|||
return dfd
|
||||
|
||||
dfd = mustbe_deferred(_append, 1)
|
||||
dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred
|
||||
steps.append(2) # add another value, that should be catched by assertEqual
|
||||
dfd.addCallback(self.assertEqual, [1, 2]) # it is [1] with maybeDeferred
|
||||
steps.append(2) # add another value, that should be catched by assertEqual
|
||||
return dfd
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -13,7 +13,7 @@ class ChunkedTest(unittest.TestCase):
|
|||
chunked_body += "8\r\n" + "sequence\r\n"
|
||||
chunked_body += "0\r\n\r\n"
|
||||
body = decode_chunked_transfer(chunked_body)
|
||||
self.assertEqual(body, \
|
||||
"This is the data in the first chunk\r\n" +
|
||||
"and this is the second one\r\n" +
|
||||
"consequence")
|
||||
self.assertEqual(body,
|
||||
"This is the data in the first chunk\r\n" +
|
||||
"and this is the second one\r\n" +
|
||||
"consequence")
|
||||
|
|
|
|||
|
|
@ -1,20 +1,16 @@
|
|||
import unittest
|
||||
|
||||
from scrapy import Spider
|
||||
from scrapy.http import Request
|
||||
from scrapy.item import BaseItem
|
||||
from scrapy.utils.spider import iterate_spider_output, iter_spider_classes
|
||||
|
||||
from scrapy.spiders import CrawlSpider
|
||||
|
||||
|
||||
class MyBaseSpider(CrawlSpider):
|
||||
pass # abstract spider
|
||||
|
||||
|
||||
class MySpider1(MyBaseSpider):
|
||||
class MySpider1(Spider):
|
||||
name = 'myspider1'
|
||||
|
||||
|
||||
class MySpider2(MyBaseSpider):
|
||||
class MySpider2(Spider):
|
||||
name = 'myspider2'
|
||||
|
||||
|
||||
|
|
@ -35,5 +31,6 @@ class UtilsSpidersTestCase(unittest.TestCase):
|
|||
it = iter_spider_classes(tests.test_utils_spider)
|
||||
self.assertEqual(set(it), {MySpider1, MySpider2})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
|
|
|||
68
tox.ini
68
tox.ini
|
|
@ -4,12 +4,12 @@
|
|||
# and then run "tox" from this directory.
|
||||
|
||||
[tox]
|
||||
envlist = py35
|
||||
envlist = security,flake8,py3
|
||||
minversion = 1.7.0
|
||||
|
||||
[testenv]
|
||||
deps =
|
||||
-ctests/constraints.txt
|
||||
-rrequirements-py3.txt
|
||||
-rtests/requirements-py3.txt
|
||||
# Extras
|
||||
botocore>=1.3.23
|
||||
|
|
@ -23,11 +23,28 @@ passenv =
|
|||
commands =
|
||||
py.test --cov=scrapy --cov-report= {posargs:--durations=10 docs scrapy tests}
|
||||
|
||||
[testenv:py35]
|
||||
basepython = python3.5
|
||||
[testenv:security]
|
||||
basepython = python3
|
||||
deps =
|
||||
bandit
|
||||
commands =
|
||||
bandit -r -c .bandit.yml {posargs:scrapy}
|
||||
|
||||
[testenv:py35-pinned]
|
||||
basepython = python3.5
|
||||
[testenv:flake8]
|
||||
basepython = python3
|
||||
deps =
|
||||
{[testenv]deps}
|
||||
pytest-flake8
|
||||
commands =
|
||||
py.test --flake8 {posargs:docs scrapy tests}
|
||||
|
||||
[testenv:pypy3]
|
||||
basepython = pypy3
|
||||
commands =
|
||||
py.test {posargs:--durations=10 docs scrapy tests}
|
||||
|
||||
[testenv:pinned]
|
||||
basepython = python3
|
||||
deps =
|
||||
-ctests/constraints.txt
|
||||
cryptography==2.0
|
||||
|
|
@ -48,34 +65,11 @@ deps =
|
|||
botocore==1.3.23
|
||||
Pillow==3.4.2
|
||||
|
||||
[testenv:py36]
|
||||
basepython = python3.6
|
||||
|
||||
[testenv:py37]
|
||||
basepython = python3.7
|
||||
|
||||
[testenv:py38]
|
||||
basepython = python3.8
|
||||
|
||||
[testenv:pypy3]
|
||||
basepython = pypy3
|
||||
commands =
|
||||
py.test {posargs:--durations=10 docs scrapy tests}
|
||||
|
||||
[testenv:security]
|
||||
basepython = python3.8
|
||||
deps =
|
||||
bandit
|
||||
commands =
|
||||
bandit -r -c .bandit.yml {posargs:scrapy}
|
||||
|
||||
[testenv:flake8]
|
||||
basepython = python3.8
|
||||
[testenv:extra-deps]
|
||||
deps =
|
||||
{[testenv]deps}
|
||||
pytest-flake8
|
||||
commands =
|
||||
py.test --flake8 {posargs:docs scrapy tests}
|
||||
reppy
|
||||
robotexclusionrulesparser
|
||||
|
||||
[docs]
|
||||
changedir = docs
|
||||
|
|
@ -83,30 +77,26 @@ deps =
|
|||
-rdocs/requirements.txt
|
||||
|
||||
[testenv:docs]
|
||||
basepython = python3
|
||||
changedir = {[docs]changedir}
|
||||
deps = {[docs]deps}
|
||||
commands =
|
||||
sphinx-build -W -b html . {envtmpdir}/html
|
||||
|
||||
[testenv:docs-coverage]
|
||||
basepython = python3
|
||||
changedir = {[docs]changedir}
|
||||
deps = {[docs]deps}
|
||||
commands =
|
||||
sphinx-build -b coverage . {envtmpdir}/coverage
|
||||
|
||||
[testenv:docs-links]
|
||||
basepython = python3
|
||||
changedir = {[docs]changedir}
|
||||
deps = {[docs]deps}
|
||||
commands =
|
||||
sphinx-build -W -b linkcheck . {envtmpdir}/linkcheck
|
||||
|
||||
[testenv:py38-extra-deps]
|
||||
basepython = python3.8
|
||||
deps =
|
||||
{[testenv]deps}
|
||||
reppy
|
||||
robotexclusionrulesparser
|
||||
|
||||
[asyncio]
|
||||
commands =
|
||||
py.test --cov=scrapy --cov-report= --reactor=asyncio {posargs:scrapy tests}
|
||||
|
|
|
|||
Loading…
Reference in New Issue