mirror of https://github.com/scrapy/scrapy.git
Merge branch 'master' into patch-1
This commit is contained in:
commit
431f6e7d90
|
|
@ -0,0 +1,18 @@
|
|||
skips:
|
||||
- B101
|
||||
- B105
|
||||
- B301
|
||||
- B303
|
||||
- B306
|
||||
- B307
|
||||
- B311
|
||||
- B320
|
||||
- B321
|
||||
- B402 # https://github.com/scrapy/scrapy/issues/4180
|
||||
- B403
|
||||
- B404
|
||||
- B406
|
||||
- B410
|
||||
- B503
|
||||
- B603
|
||||
- B605
|
||||
|
|
@ -0,0 +1,11 @@
|
|||
version: 2
|
||||
sphinx:
|
||||
configuration: docs/conf.py
|
||||
fail_on_warning: true
|
||||
python:
|
||||
# For available versions, see:
|
||||
# https://docs.readthedocs.io/en/stable/config-file/v2.html#build-image
|
||||
version: 3.7 # Keep in sync with .travis.yml
|
||||
install:
|
||||
- requirements: docs/requirements.txt
|
||||
- path: .
|
||||
32
.travis.yml
32
.travis.yml
|
|
@ -7,39 +7,31 @@ branches:
|
|||
- /^\d\.\d+\.\d+(rc\d+|\.dev\d+)?$/
|
||||
matrix:
|
||||
include:
|
||||
- env: TOXENV=py27
|
||||
python: 2.7
|
||||
- env: TOXENV=py27-pinned
|
||||
python: 2.7
|
||||
- env: TOXENV=py27-extra-deps
|
||||
python: 2.7
|
||||
- env: TOXENV=pypy
|
||||
python: 2.7
|
||||
- env: TOXENV=security
|
||||
python: 3.8
|
||||
- env: TOXENV=flake8
|
||||
python: 3.8
|
||||
- env: TOXENV=pypy3
|
||||
python: 3.5
|
||||
- env: TOXENV=py35
|
||||
python: 3.5
|
||||
- env: TOXENV=py35-pinned
|
||||
- env: TOXENV=pinned
|
||||
python: 3.5
|
||||
- env: TOXENV=py35-asyncio
|
||||
python: 3.5.2
|
||||
- env: TOXENV=py36
|
||||
python: 3.6
|
||||
- env: TOXENV=py37
|
||||
python: 3.7
|
||||
- env: TOXENV=py38
|
||||
python: 3.8
|
||||
- env: TOXENV=py38-extra-deps
|
||||
- env: TOXENV=extra-deps
|
||||
python: 3.8
|
||||
- env: TOXENV=py38-asyncio
|
||||
python: 3.8
|
||||
- env: TOXENV=docs
|
||||
python: 3.6
|
||||
python: 3.7 # Keep in sync with .readthedocs.yml
|
||||
install:
|
||||
- |
|
||||
if [ "$TOXENV" = "pypy" ]; then
|
||||
export PYPY_VERSION="pypy-6.0.0-linux_x86_64-portable"
|
||||
wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2"
|
||||
tar -jxf ${PYPY_VERSION}.tar.bz2
|
||||
virtualenv --python="$PYPY_VERSION/bin/pypy" "$HOME/virtualenvs/$PYPY_VERSION"
|
||||
source "$HOME/virtualenvs/$PYPY_VERSION/bin/activate"
|
||||
fi
|
||||
if [ "$TOXENV" = "pypy3" ]; then
|
||||
export PYPY_VERSION="pypy3.5-5.9-beta-linux_x86_64-portable"
|
||||
wget "https://bitbucket.org/squeaky/portable-pypy/downloads/${PYPY_VERSION}.tar.bz2"
|
||||
|
|
@ -70,4 +62,4 @@ deploy:
|
|||
on:
|
||||
tags: true
|
||||
repo: scrapy/scrapy
|
||||
condition: "$TOXENV == py27 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$"
|
||||
condition: "$TOXENV == py37 && $TRAVIS_TAG =~ ^[0-9]+[.][0-9]+[.][0-9]+(rc[0-9]+|[.]dev[0-9]+)?$"
|
||||
|
|
|
|||
|
|
@ -68,7 +68,7 @@ members of the project's leadership.
|
|||
## Attribution
|
||||
|
||||
This Code of Conduct is adapted from the [Contributor Covenant][homepage], version 1.4,
|
||||
available at [http://contributor-covenant.org/version/1/4][version]
|
||||
available at [http://contributor-covenant.org/version/1/4][version].
|
||||
|
||||
[homepage]: http://contributor-covenant.org
|
||||
[version]: http://contributor-covenant.org/version/1/4/
|
||||
|
|
|
|||
22
README.rst
22
README.rst
|
|
@ -34,14 +34,14 @@ Scrapy is a fast high-level web crawling and web scraping framework, used to
|
|||
crawl websites and extract structured data from their pages. It can be used for
|
||||
a wide range of purposes, from data mining to monitoring and automated testing.
|
||||
|
||||
For more information including a list of features check the Scrapy homepage at:
|
||||
https://scrapy.org
|
||||
Check the Scrapy homepage at https://scrapy.org for more information,
|
||||
including a list of features.
|
||||
|
||||
Requirements
|
||||
============
|
||||
|
||||
* Python 2.7 or Python 3.5+
|
||||
* Works on Linux, Windows, Mac OSX, BSD
|
||||
* Python 3.5+
|
||||
* Works on Linux, Windows, macOS, BSD
|
||||
|
||||
Install
|
||||
=======
|
||||
|
|
@ -50,8 +50,8 @@ The quick way::
|
|||
|
||||
pip install scrapy
|
||||
|
||||
For more details see the install section in the documentation:
|
||||
https://docs.scrapy.org/en/latest/intro/install.html
|
||||
See the install section in the documentation at
|
||||
https://docs.scrapy.org/en/latest/intro/install.html for more details.
|
||||
|
||||
Documentation
|
||||
=============
|
||||
|
|
@ -62,17 +62,17 @@ directory.
|
|||
Releases
|
||||
========
|
||||
|
||||
You can find release notes at https://docs.scrapy.org/en/latest/news.html
|
||||
You can check https://docs.scrapy.org/en/latest/news.html for the release notes.
|
||||
|
||||
Community (blog, twitter, mail list, IRC)
|
||||
=========================================
|
||||
|
||||
See https://scrapy.org/community/
|
||||
See https://scrapy.org/community/ for details.
|
||||
|
||||
Contributing
|
||||
============
|
||||
|
||||
See https://docs.scrapy.org/en/master/contributing.html
|
||||
See https://docs.scrapy.org/en/master/contributing.html for details.
|
||||
|
||||
Code of Conduct
|
||||
---------------
|
||||
|
|
@ -86,9 +86,9 @@ Please report unacceptable behavior to opensource@scrapinghub.com.
|
|||
Companies using Scrapy
|
||||
======================
|
||||
|
||||
See https://scrapy.org/companies/
|
||||
See https://scrapy.org/companies/ for a list.
|
||||
|
||||
Commercial Support
|
||||
==================
|
||||
|
||||
See https://scrapy.org/support/
|
||||
See https://scrapy.org/support/ for details.
|
||||
|
|
|
|||
|
|
@ -1,5 +1,4 @@
|
|||
:orphan:
|
||||
|
||||
==============
|
||||
Scrapy artwork
|
||||
==============
|
||||
|
||||
|
|
|
|||
46
conftest.py
46
conftest.py
|
|
@ -1,21 +1,53 @@
|
|||
import six
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
def _py_files(folder):
|
||||
return (str(p) for p in Path(folder).rglob('*.py'))
|
||||
|
||||
|
||||
collect_ignore = [
|
||||
# not a test, but looks like a test
|
||||
"scrapy/utils/testsite.py",
|
||||
# contains scripts to be run by tests/test_crawler.py::CrawlerProcessSubprocess
|
||||
*_py_files("tests/CrawlerProcess"),
|
||||
# Py36-only parts of respective tests
|
||||
*_py_files("tests/py36"),
|
||||
]
|
||||
|
||||
|
||||
if six.PY3:
|
||||
for line in open('tests/py3-ignores.txt'):
|
||||
file_path = line.strip()
|
||||
if file_path and file_path[0] != '#':
|
||||
collect_ignore.append(file_path)
|
||||
for line in open('tests/ignores.txt'):
|
||||
file_path = line.strip()
|
||||
if file_path and file_path[0] != '#':
|
||||
collect_ignore.append(file_path)
|
||||
|
||||
|
||||
@pytest.fixture()
|
||||
def chdir(tmpdir):
|
||||
"""Change to pytest-provided temporary directory"""
|
||||
tmpdir.chdir()
|
||||
|
||||
|
||||
def pytest_collection_modifyitems(session, config, items):
|
||||
# Avoid executing tests when executing `--flake8` flag (pytest-flake8)
|
||||
try:
|
||||
from pytest_flake8 import Flake8Item
|
||||
if config.getoption('--flake8'):
|
||||
items[:] = [item for item in items if isinstance(item, Flake8Item)]
|
||||
except ImportError:
|
||||
pass
|
||||
|
||||
|
||||
@pytest.fixture(scope='class')
|
||||
def reactor_pytest(request):
|
||||
if not request.cls:
|
||||
# doctests
|
||||
return
|
||||
request.cls.reactor_pytest = request.config.getoption("--reactor")
|
||||
return request.cls.reactor_pytest
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def only_asyncio(request, reactor_pytest):
|
||||
if request.node.get_closest_marker('only_asyncio') and reactor_pytest != 'asyncio':
|
||||
pytest.skip('This test is only run with --reactor=asyncio')
|
||||
|
|
|
|||
|
|
@ -0,0 +1,281 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>Quotes to Scrape</title>
|
||||
<link rel="stylesheet" href="/static/bootstrap.min.css">
|
||||
<link rel="stylesheet" href="/static/main.css">
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<div class="row header-box">
|
||||
<div class="col-md-8">
|
||||
<h1>
|
||||
<a href="/" style="text-decoration: none">Quotes to Scrape</a>
|
||||
</h1>
|
||||
</div>
|
||||
<div class="col-md-4">
|
||||
<p>
|
||||
|
||||
<a href="/login">Login</a>
|
||||
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
|
||||
<div class="row">
|
||||
<div class="col-md-8">
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
|
||||
|
||||
<a class="tag" href="/tag/change/page/1/">change</a>
|
||||
|
||||
<a class="tag" href="/tag/deep-thoughts/page/1/">deep-thoughts</a>
|
||||
|
||||
<a class="tag" href="/tag/thinking/page/1/">thinking</a>
|
||||
|
||||
<a class="tag" href="/tag/world/page/1/">world</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“It is our choices, Harry, that show what we truly are, far more than our abilities.”</span>
|
||||
<span>by <small class="author" itemprop="author">J.K. Rowling</small>
|
||||
<a href="/author/J-K-Rowling">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
|
||||
|
||||
<a class="tag" href="/tag/abilities/page/1/">abilities</a>
|
||||
|
||||
<a class="tag" href="/tag/choices/page/1/">choices</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
<a class="tag" href="/tag/life/page/1/">life</a>
|
||||
|
||||
<a class="tag" href="/tag/live/page/1/">live</a>
|
||||
|
||||
<a class="tag" href="/tag/miracle/page/1/">miracle</a>
|
||||
|
||||
<a class="tag" href="/tag/miracles/page/1/">miracles</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.”</span>
|
||||
<span>by <small class="author" itemprop="author">Jane Austen</small>
|
||||
<a href="/author/Jane-Austen">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
|
||||
|
||||
<a class="tag" href="/tag/aliteracy/page/1/">aliteracy</a>
|
||||
|
||||
<a class="tag" href="/tag/books/page/1/">books</a>
|
||||
|
||||
<a class="tag" href="/tag/classic/page/1/">classic</a>
|
||||
|
||||
<a class="tag" href="/tag/humor/page/1/">humor</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.”</span>
|
||||
<span>by <small class="author" itemprop="author">Marilyn Monroe</small>
|
||||
<a href="/author/Marilyn-Monroe">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
|
||||
|
||||
<a class="tag" href="/tag/be-yourself/page/1/">be-yourself</a>
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“Try not to become a man of success. Rather become a man of value.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
|
||||
|
||||
<a class="tag" href="/tag/adulthood/page/1/">adulthood</a>
|
||||
|
||||
<a class="tag" href="/tag/success/page/1/">success</a>
|
||||
|
||||
<a class="tag" href="/tag/value/page/1/">value</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“It is better to be hated for what you are than to be loved for what you are not.”</span>
|
||||
<span>by <small class="author" itemprop="author">André Gide</small>
|
||||
<a href="/author/Andre-Gide">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="life,love" / >
|
||||
|
||||
<a class="tag" href="/tag/life/page/1/">life</a>
|
||||
|
||||
<a class="tag" href="/tag/love/page/1/">love</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“I have not failed. I've just found 10,000 ways that won't work.”</span>
|
||||
<span>by <small class="author" itemprop="author">Thomas A. Edison</small>
|
||||
<a href="/author/Thomas-A-Edison">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
|
||||
|
||||
<a class="tag" href="/tag/edison/page/1/">edison</a>
|
||||
|
||||
<a class="tag" href="/tag/failure/page/1/">failure</a>
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
<a class="tag" href="/tag/paraphrased/page/1/">paraphrased</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“A woman is like a tea bag; you never know how strong it is until it's in hot water.”</span>
|
||||
<span>by <small class="author" itemprop="author">Eleanor Roosevelt</small>
|
||||
<a href="/author/Eleanor-Roosevelt">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
|
||||
|
||||
<a class="tag" href="/tag/misattributed-eleanor-roosevelt/page/1/">misattributed-eleanor-roosevelt</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“A day without sunshine is like, you know, night.”</span>
|
||||
<span>by <small class="author" itemprop="author">Steve Martin</small>
|
||||
<a href="/author/Steve-Martin">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
|
||||
|
||||
<a class="tag" href="/tag/humor/page/1/">humor</a>
|
||||
|
||||
<a class="tag" href="/tag/obvious/page/1/">obvious</a>
|
||||
|
||||
<a class="tag" href="/tag/simile/page/1/">simile</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<nav>
|
||||
<ul class="pager">
|
||||
|
||||
|
||||
<li class="next">
|
||||
<a href="/page/2/">Next <span aria-hidden="true">→</span></a>
|
||||
</li>
|
||||
|
||||
</ul>
|
||||
</nav>
|
||||
</div>
|
||||
<div class="col-md-4 tags-box">
|
||||
|
||||
<h2>Top Ten tags</h2>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 28px" href="/tag/love/">love</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 26px" href="/tag/inspirational/">inspirational</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 26px" href="/tag/life/">life</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 24px" href="/tag/humor/">humor</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 22px" href="/tag/books/">books</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 14px" href="/tag/reading/">reading</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 10px" href="/tag/friendship/">friendship</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 8px" href="/tag/friends/">friends</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 8px" href="/tag/truth/">truth</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 6px" href="/tag/simile/">simile</a>
|
||||
</span>
|
||||
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
<footer class="footer">
|
||||
<div class="container">
|
||||
<p class="text-muted">
|
||||
Quotes by: <a href="https://www.goodreads.com/quotes">GoodReads.com</a>
|
||||
</p>
|
||||
<p class="copyright">
|
||||
Made with <span class='sh-red'>❤</span> by <a href="https://scrapinghub.com">Scrapinghub</a>
|
||||
</p>
|
||||
</div>
|
||||
</footer>
|
||||
</body>
|
||||
</html>
|
||||
|
|
@ -0,0 +1,281 @@
|
|||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8">
|
||||
<title>Quotes to Scrape</title>
|
||||
<link rel="stylesheet" href="/static/bootstrap.min.css">
|
||||
<link rel="stylesheet" href="/static/main.css">
|
||||
</head>
|
||||
<body>
|
||||
<div class="container">
|
||||
<div class="row header-box">
|
||||
<div class="col-md-8">
|
||||
<h1>
|
||||
<a href="/" style="text-decoration: none">Quotes to Scrape</a>
|
||||
</h1>
|
||||
</div>
|
||||
<div class="col-md-4">
|
||||
<p>
|
||||
|
||||
<a href="/login">Login</a>
|
||||
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
|
||||
<div class="row">
|
||||
<div class="col-md-8">
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
|
||||
|
||||
<a class="tag" href="/tag/change/page/1/">change</a>
|
||||
|
||||
<a class="tag" href="/tag/deep-thoughts/page/1/">deep-thoughts</a>
|
||||
|
||||
<a class="tag" href="/tag/thinking/page/1/">thinking</a>
|
||||
|
||||
<a class="tag" href="/tag/world/page/1/">world</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“It is our choices, Harry, that show what we truly are, far more than our abilities.”</span>
|
||||
<span>by <small class="author" itemprop="author">J.K. Rowling</small>
|
||||
<a href="/author/J-K-Rowling">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
|
||||
|
||||
<a class="tag" href="/tag/abilities/page/1/">abilities</a>
|
||||
|
||||
<a class="tag" href="/tag/choices/page/1/">choices</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
<a class="tag" href="/tag/life/page/1/">life</a>
|
||||
|
||||
<a class="tag" href="/tag/live/page/1/">live</a>
|
||||
|
||||
<a class="tag" href="/tag/miracle/page/1/">miracle</a>
|
||||
|
||||
<a class="tag" href="/tag/miracles/page/1/">miracles</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“The person, be it gentleman or lady, who has not pleasure in a good novel, must be intolerably stupid.”</span>
|
||||
<span>by <small class="author" itemprop="author">Jane Austen</small>
|
||||
<a href="/author/Jane-Austen">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
|
||||
|
||||
<a class="tag" href="/tag/aliteracy/page/1/">aliteracy</a>
|
||||
|
||||
<a class="tag" href="/tag/books/page/1/">books</a>
|
||||
|
||||
<a class="tag" href="/tag/classic/page/1/">classic</a>
|
||||
|
||||
<a class="tag" href="/tag/humor/page/1/">humor</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“Imperfection is beauty, madness is genius and it's better to be absolutely ridiculous than absolutely boring.”</span>
|
||||
<span>by <small class="author" itemprop="author">Marilyn Monroe</small>
|
||||
<a href="/author/Marilyn-Monroe">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
|
||||
|
||||
<a class="tag" href="/tag/be-yourself/page/1/">be-yourself</a>
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“Try not to become a man of success. Rather become a man of value.”</span>
|
||||
<span>by <small class="author" itemprop="author">Albert Einstein</small>
|
||||
<a href="/author/Albert-Einstein">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
|
||||
|
||||
<a class="tag" href="/tag/adulthood/page/1/">adulthood</a>
|
||||
|
||||
<a class="tag" href="/tag/success/page/1/">success</a>
|
||||
|
||||
<a class="tag" href="/tag/value/page/1/">value</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“It is better to be hated for what you are than to be loved for what you are not.”</span>
|
||||
<span>by <small class="author" itemprop="author">André Gide</small>
|
||||
<a href="/author/Andre-Gide">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="life,love" / >
|
||||
|
||||
<a class="tag" href="/tag/life/page/1/">life</a>
|
||||
|
||||
<a class="tag" href="/tag/love/page/1/">love</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“I have not failed. I've just found 10,000 ways that won't work.”</span>
|
||||
<span>by <small class="author" itemprop="author">Thomas A. Edison</small>
|
||||
<a href="/author/Thomas-A-Edison">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
|
||||
|
||||
<a class="tag" href="/tag/edison/page/1/">edison</a>
|
||||
|
||||
<a class="tag" href="/tag/failure/page/1/">failure</a>
|
||||
|
||||
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
|
||||
|
||||
<a class="tag" href="/tag/paraphrased/page/1/">paraphrased</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“A woman is like a tea bag; you never know how strong it is until it's in hot water.”</span>
|
||||
<span>by <small class="author" itemprop="author">Eleanor Roosevelt</small>
|
||||
<a href="/author/Eleanor-Roosevelt">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
|
||||
|
||||
<a class="tag" href="/tag/misattributed-eleanor-roosevelt/page/1/">misattributed-eleanor-roosevelt</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div class="quote" itemscope itemtype="http://schema.org/CreativeWork">
|
||||
<span class="text" itemprop="text">“A day without sunshine is like, you know, night.”</span>
|
||||
<span>by <small class="author" itemprop="author">Steve Martin</small>
|
||||
<a href="/author/Steve-Martin">(about)</a>
|
||||
</span>
|
||||
<div class="tags">
|
||||
Tags:
|
||||
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
|
||||
|
||||
<a class="tag" href="/tag/humor/page/1/">humor</a>
|
||||
|
||||
<a class="tag" href="/tag/obvious/page/1/">obvious</a>
|
||||
|
||||
<a class="tag" href="/tag/simile/page/1/">simile</a>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<nav>
|
||||
<ul class="pager">
|
||||
|
||||
|
||||
<li class="next">
|
||||
<a href="/page/2/">Next <span aria-hidden="true">→</span></a>
|
||||
</li>
|
||||
|
||||
</ul>
|
||||
</nav>
|
||||
</div>
|
||||
<div class="col-md-4 tags-box">
|
||||
|
||||
<h2>Top Ten tags</h2>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 28px" href="/tag/love/">love</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 26px" href="/tag/inspirational/">inspirational</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 26px" href="/tag/life/">life</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 24px" href="/tag/humor/">humor</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 22px" href="/tag/books/">books</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 14px" href="/tag/reading/">reading</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 10px" href="/tag/friendship/">friendship</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 8px" href="/tag/friends/">friends</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 8px" href="/tag/truth/">truth</a>
|
||||
</span>
|
||||
|
||||
<span class="tag-item">
|
||||
<a class="tag" style="font-size: 6px" href="/tag/simile/">simile</a>
|
||||
</span>
|
||||
|
||||
|
||||
</div>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
<footer class="footer">
|
||||
<div class="container">
|
||||
<p class="text-muted">
|
||||
Quotes by: <a href="https://www.goodreads.com/quotes">GoodReads.com</a>
|
||||
</p>
|
||||
<p class="copyright">
|
||||
Made with <span class='sh-red'>❤</span> by <a href="https://scrapinghub.com">Scrapinghub</a>
|
||||
</p>
|
||||
</div>
|
||||
</footer>
|
||||
</body>
|
||||
</html>
|
||||
31
docs/conf.py
31
docs/conf.py
|
|
@ -12,6 +12,7 @@
|
|||
# serve to show the default.
|
||||
|
||||
import sys
|
||||
from datetime import datetime
|
||||
from os import path
|
||||
|
||||
# If your extensions are in another directory, add it here. If the directory
|
||||
|
|
@ -27,10 +28,13 @@ sys.path.insert(0, path.dirname(path.dirname(__file__)))
|
|||
# Add any Sphinx extension module names here, as strings. They can be extensions
|
||||
# coming with Sphinx (named 'sphinx.ext.*') or your custom ones.
|
||||
extensions = [
|
||||
'hoverxref.extension',
|
||||
'notfound.extension',
|
||||
'scrapydocs',
|
||||
'sphinx.ext.autodoc',
|
||||
'sphinx.ext.coverage',
|
||||
'sphinx.ext.intersphinx',
|
||||
'sphinx.ext.viewcode',
|
||||
]
|
||||
|
||||
# Add any paths that contain templates here, relative to this directory.
|
||||
|
|
@ -46,8 +50,8 @@ source_suffix = '.rst'
|
|||
master_doc = 'index'
|
||||
|
||||
# General information about the project.
|
||||
project = u'Scrapy'
|
||||
copyright = u'2008–2018, Scrapy developers'
|
||||
project = 'Scrapy'
|
||||
copyright = '2008–{}, Scrapy developers'.format(datetime.now().year)
|
||||
|
||||
# The version info for the project you're documenting, acts as replacement for
|
||||
# |version| and |release|, also used in various other places throughout the
|
||||
|
|
@ -191,8 +195,8 @@ htmlhelp_basename = 'Scrapydoc'
|
|||
# Grouping the document tree into LaTeX files. List of tuples
|
||||
# (source start file, target name, title, author, document class [howto/manual]).
|
||||
latex_documents = [
|
||||
('index', 'Scrapy.tex', u'Scrapy Documentation',
|
||||
u'Scrapy developers', 'manual'),
|
||||
('index', 'Scrapy.tex', 'Scrapy Documentation',
|
||||
'Scrapy developers', 'manual'),
|
||||
]
|
||||
|
||||
# The name of an image file (relative to this directory) to place at the top of
|
||||
|
|
@ -237,7 +241,7 @@ coverage_ignore_pyobjects = [
|
|||
r'\bContractsManager\b$',
|
||||
|
||||
# For default contracts we only want to document their general purpose in
|
||||
# their constructor, the methods they reimplement to achieve that purpose
|
||||
# their __init__ method, the methods they reimplement to achieve that purpose
|
||||
# should be irrelevant to developers using those contracts.
|
||||
r'\w+Contract\.(adjust_request_args|(pre|post)_process)$',
|
||||
|
||||
|
|
@ -265,6 +269,10 @@ coverage_ignore_pyobjects = [
|
|||
|
||||
# Never documented before, and deprecated now.
|
||||
r'^scrapy\.item\.DictItem$',
|
||||
r'^scrapy\.linkextractors\.FilteringLinkExtractor$',
|
||||
|
||||
# Implementation detail of LxmlLinkExtractor
|
||||
r'^scrapy\.linkextractors\.lxmlhtml\.LxmlParserLinkExtractor',
|
||||
]
|
||||
|
||||
|
||||
|
|
@ -272,5 +280,18 @@ coverage_ignore_pyobjects = [
|
|||
# -------------------------------------
|
||||
|
||||
intersphinx_mapping = {
|
||||
'coverage': ('https://coverage.readthedocs.io/en/stable', None),
|
||||
'cssselect': ('https://cssselect.readthedocs.io/en/latest', None),
|
||||
'pytest': ('https://docs.pytest.org/en/latest', None),
|
||||
'python': ('https://docs.python.org/3', None),
|
||||
'sphinx': ('https://www.sphinx-doc.org/en/master', None),
|
||||
'tox': ('https://tox.readthedocs.io/en/latest', None),
|
||||
'twisted': ('https://twistedmatrix.com/documents/current', None),
|
||||
'twistedapi': ('https://twistedmatrix.com/documents/current/api', None),
|
||||
}
|
||||
|
||||
|
||||
# Options for sphinx-hoverxref options
|
||||
# ------------------------------------
|
||||
|
||||
hoverxref_auto_ref = True
|
||||
|
|
|
|||
|
|
@ -0,0 +1,29 @@
|
|||
import os
|
||||
from doctest import ELLIPSIS, NORMALIZE_WHITESPACE
|
||||
|
||||
from scrapy.http.response.html import HtmlResponse
|
||||
from sybil import Sybil
|
||||
from sybil.parsers.codeblock import CodeBlockParser
|
||||
from sybil.parsers.doctest import DocTestParser
|
||||
from sybil.parsers.skip import skip
|
||||
|
||||
|
||||
def load_response(url, filename):
|
||||
input_path = os.path.join(os.path.dirname(__file__), '_tests', filename)
|
||||
with open(input_path, 'rb') as input_file:
|
||||
return HtmlResponse(url, body=input_file.read())
|
||||
|
||||
|
||||
def setup(namespace):
|
||||
namespace['load_response'] = load_response
|
||||
|
||||
|
||||
pytest_collect_file = Sybil(
|
||||
parsers=[
|
||||
DocTestParser(optionflags=ELLIPSIS | NORMALIZE_WHITESPACE),
|
||||
CodeBlockParser(future_imports=['print_function']),
|
||||
skip,
|
||||
],
|
||||
pattern='*.rst',
|
||||
setup=setup,
|
||||
).pytest()
|
||||
|
|
@ -44,7 +44,7 @@ guidelines when you're going to report a new bug.
|
|||
* check the :ref:`FAQ <faq>` first to see if your issue is addressed in a
|
||||
well-known question
|
||||
|
||||
* if you have a general question about scrapy usage, please ask it at
|
||||
* if you have a general question about Scrapy usage, please ask it at
|
||||
`Stack Overflow <https://stackoverflow.com/questions/tagged/scrapy>`__
|
||||
(use "scrapy" tag).
|
||||
|
||||
|
|
@ -143,7 +143,7 @@ by running ``git fetch upstream pull/$PR_NUMBER/head:$BRANCH_NAME_TO_CREATE``
|
|||
(replace 'upstream' with a remote name for scrapy repository,
|
||||
``$PR_NUMBER`` with an ID of the pull request, and ``$BRANCH_NAME_TO_CREATE``
|
||||
with a name of the branch you want to create locally).
|
||||
See also: https://help.github.com/articles/checking-out-pull-requests-locally/#modifying-an-inactive-pull-request-locally.
|
||||
See also: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/checking-out-pull-requests-locally#modifying-an-inactive-pull-request-locally.
|
||||
|
||||
When writing GitHub pull requests, try to keep titles short but descriptive.
|
||||
E.g. For bug #411: "Scrapy hangs if an exception raises in start_requests"
|
||||
|
|
@ -168,7 +168,7 @@ Scrapy:
|
|||
|
||||
* Don't put your name in the code you contribute; git provides enough
|
||||
metadata to identify author of the code.
|
||||
See https://help.github.com/articles/setting-your-username-in-git/ for
|
||||
See https://help.github.com/en/github/using-git/setting-your-username-in-git for
|
||||
setup instructions.
|
||||
|
||||
.. _documentation-policies:
|
||||
|
|
@ -177,79 +177,71 @@ Documentation policies
|
|||
======================
|
||||
|
||||
For reference documentation of API members (classes, methods, etc.) use
|
||||
docstrings and make sure that the Sphinx documentation uses the autodoc_
|
||||
extension to pull the docstrings. API reference documentation should follow
|
||||
docstring conventions (`PEP 257`_) and be IDE-friendly: short, to the point,
|
||||
and it may provide short examples.
|
||||
docstrings and make sure that the Sphinx documentation uses the
|
||||
:mod:`~sphinx.ext.autodoc` extension to pull the docstrings. API reference
|
||||
documentation should follow docstring conventions (`PEP 257`_) and be
|
||||
IDE-friendly: short, to the point, and it may provide short examples.
|
||||
|
||||
Other types of documentation, such as tutorials or topics, should be covered in
|
||||
files within the ``docs/`` directory. This includes documentation that is
|
||||
specific to an API member, but goes beyond API reference documentation.
|
||||
|
||||
In any case, if something is covered in a docstring, use the autodoc_
|
||||
extension to pull the docstring into the documentation instead of duplicating
|
||||
the docstring in files within the ``docs/`` directory.
|
||||
|
||||
.. _autodoc: http://www.sphinx-doc.org/en/stable/ext/autodoc.html
|
||||
In any case, if something is covered in a docstring, use the
|
||||
:mod:`~sphinx.ext.autodoc` extension to pull the docstring into the
|
||||
documentation instead of duplicating the docstring in files within the
|
||||
``docs/`` directory.
|
||||
|
||||
Tests
|
||||
=====
|
||||
|
||||
Tests are implemented using the `Twisted unit-testing framework`_, running
|
||||
tests requires `tox`_.
|
||||
Tests are implemented using the :doc:`Twisted unit-testing framework
|
||||
<twisted:core/development/policy/test-standard>`. Running tests requires
|
||||
:doc:`tox <tox:index>`.
|
||||
|
||||
.. _running-tests:
|
||||
|
||||
Running tests
|
||||
-------------
|
||||
|
||||
Make sure you have a recent enough `tox`_ installation:
|
||||
To run all tests::
|
||||
|
||||
``tox --version``
|
||||
|
||||
If your version is older than 1.7.0, please update it first:
|
||||
|
||||
``pip install -U tox``
|
||||
|
||||
To run all tests go to the root directory of Scrapy source code and run:
|
||||
|
||||
``tox``
|
||||
tox
|
||||
|
||||
To run a specific test (say ``tests/test_loader.py``) use:
|
||||
|
||||
``tox -- tests/test_loader.py``
|
||||
|
||||
To run the tests on a specific tox_ environment, use ``-e <name>`` with an
|
||||
environment name from ``tox.ini``. For example, to run the tests with Python
|
||||
3.6 use::
|
||||
To run the tests on a specific :doc:`tox <tox:index>` environment, use
|
||||
``-e <name>`` with an environment name from ``tox.ini``. For example, to run
|
||||
the tests with Python 3.6 use::
|
||||
|
||||
tox -e py36
|
||||
|
||||
You can also specify a comma-separated list of environmets, and use `tox’s
|
||||
parallel mode`_ to run the tests on multiple environments in parallel::
|
||||
You can also specify a comma-separated list of environments, and use :ref:`tox’s
|
||||
parallel mode <tox:parallel_mode>` to run the tests on multiple environments in
|
||||
parallel::
|
||||
|
||||
tox -e py27,py36 -p auto
|
||||
tox -e py36,py38 -p auto
|
||||
|
||||
To pass command-line options to pytest_, add them after ``--`` in your call to
|
||||
tox_. Using ``--`` overrides the default positional arguments defined in
|
||||
``tox.ini``, so you must include those default positional arguments
|
||||
(``scrapy tests``) after ``--`` as well::
|
||||
To pass command-line options to :doc:`pytest <pytest:index>`, add them after
|
||||
``--`` in your call to :doc:`tox <tox:index>`. Using ``--`` overrides the
|
||||
default positional arguments defined in ``tox.ini``, so you must include those
|
||||
default positional arguments (``scrapy tests``) after ``--`` as well::
|
||||
|
||||
tox -- scrapy tests -x # stop after first failure
|
||||
|
||||
You can also use the `pytest-xdist`_ plugin. For example, to run all tests on
|
||||
the Python 3.6 tox_ environment using all your CPU cores::
|
||||
the Python 3.6 :doc:`tox <tox:index>` environment using all your CPU cores::
|
||||
|
||||
tox -e py36 -- scrapy tests -n auto
|
||||
|
||||
To see coverage report install `coverage`_ (``pip install coverage``) and run:
|
||||
To see coverage report install :doc:`coverage <coverage:index>`
|
||||
(``pip install coverage``) and run:
|
||||
|
||||
``coverage report``
|
||||
|
||||
see output of ``coverage --help`` for more options like html or xml report.
|
||||
|
||||
.. _coverage: https://pypi.python.org/pypi/coverage
|
||||
|
||||
Writing tests
|
||||
-------------
|
||||
|
||||
|
|
@ -270,13 +262,9 @@ And their unit-tests are in::
|
|||
.. _issue tracker: https://github.com/scrapy/scrapy/issues
|
||||
.. _scrapy-users: https://groups.google.com/forum/#!forum/scrapy-users
|
||||
.. _Scrapy subreddit: https://reddit.com/r/scrapy
|
||||
.. _Twisted unit-testing framework: https://twistedmatrix.com/documents/current/core/development/policy/test-standard.html
|
||||
.. _AUTHORS: https://github.com/scrapy/scrapy/blob/master/AUTHORS
|
||||
.. _tests/: https://github.com/scrapy/scrapy/tree/master/tests
|
||||
.. _open issues: https://github.com/scrapy/scrapy/issues
|
||||
.. _PEP 257: https://www.python.org/dev/peps/pep-0257/
|
||||
.. _pull request: https://help.github.com/en/articles/creating-a-pull-request
|
||||
.. _pytest: https://docs.pytest.org/en/latest/usage.html
|
||||
.. _pytest-xdist: https://docs.pytest.org/en/3.0.0/xdist.html
|
||||
.. _tox: https://pypi.python.org/pypi/tox
|
||||
.. _tox’s parallel mode: https://tox.readthedocs.io/en/latest/example/basic.html#parallel-mode
|
||||
.. _pull request: https://help.github.com/en/github/collaborating-with-issues-and-pull-requests/creating-a-pull-request
|
||||
.. _pytest-xdist: https://github.com/pytest-dev/pytest-xdist
|
||||
|
|
|
|||
32
docs/faq.rst
32
docs/faq.rst
|
|
@ -22,8 +22,8 @@ In other words, comparing `BeautifulSoup`_ (or `lxml`_) to Scrapy is like
|
|||
comparing `jinja2`_ to `Django`_.
|
||||
|
||||
.. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _jinja2: http://jinja.pocoo.org/
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _jinja2: https://palletsprojects.com/p/jinja/
|
||||
.. _Django: https://www.djangoproject.com/
|
||||
|
||||
Can I use Scrapy with BeautifulSoup?
|
||||
|
|
@ -69,11 +69,11 @@ Here's an example spider using BeautifulSoup API, with ``lxml`` as the HTML pars
|
|||
What Python versions does Scrapy support?
|
||||
-----------------------------------------
|
||||
|
||||
Scrapy is supported under Python 2.7 and Python 3.5+
|
||||
Scrapy is supported under Python 3.5+
|
||||
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
|
||||
Python 2.6 support was dropped starting at Scrapy 0.20.
|
||||
Python 3 support was added in Scrapy 1.1.
|
||||
PyPy support was added in Scrapy 1.4, PyPy3 support was added in Scrapy 1.5.
|
||||
Python 2 support was dropped in Scrapy 2.0.
|
||||
|
||||
.. note::
|
||||
For Python 3 support on Windows, it is recommended to use
|
||||
|
|
@ -140,7 +140,7 @@ setting the following settings::
|
|||
|
||||
While pending requests are below the configured values of
|
||||
:setting:`CONCURRENT_REQUESTS`, :setting:`CONCURRENT_REQUESTS_PER_DOMAIN` or
|
||||
:setting:`CONCURRENT_REQUESTS_PER_DOMAIN`, those requests are sent
|
||||
:setting:`CONCURRENT_REQUESTS_PER_IP`, those requests are sent
|
||||
concurrently. As a result, the first few requests of a crawl rarely follow the
|
||||
desired order. Lowering those settings to ``1`` enforces the desired order, but
|
||||
it significantly slows down the crawl as a whole.
|
||||
|
|
@ -269,7 +269,7 @@ The ``__VIEWSTATE`` parameter is used in sites built with ASP.NET/VB.NET. For
|
|||
more info on how it works see `this page`_. Also, here's an `example spider`_
|
||||
which scrapes one of these sites.
|
||||
|
||||
.. _this page: http://search.cpan.org/~ecarroll/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm
|
||||
.. _this page: https://metacpan.org/pod/release/ECARROLL/HTML-TreeBuilderX-ASP_NET-0.09/lib/HTML/TreeBuilderX/ASP_NET.pm
|
||||
.. _example spider: https://github.com/AmbientLighter/rpn-fas/blob/master/fas/spiders/rnp.py
|
||||
|
||||
What's the best way to parse big XML/CSV data feeds?
|
||||
|
|
@ -338,7 +338,7 @@ How to split an item into multiple items in an item pipeline?
|
|||
input item. :ref:`Create a spider middleware <custom-spider-middleware>`
|
||||
instead, and use its
|
||||
:meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_spider_output`
|
||||
method for this puspose. For example::
|
||||
method for this purpose. For example::
|
||||
|
||||
from copy import deepcopy
|
||||
|
||||
|
|
@ -353,7 +353,25 @@ method for this puspose. For example::
|
|||
for _ in range(item['multiply_by']):
|
||||
yield deepcopy(item)
|
||||
|
||||
Does Scrapy support IPv6 addresses?
|
||||
-----------------------------------
|
||||
|
||||
Yes, by setting :setting:`DNS_RESOLVER` to ``scrapy.resolver.CachingHostnameResolver``.
|
||||
Note that by doing so, you lose the ability to set a specific timeout for DNS requests
|
||||
(the value of the :setting:`DNS_TIMEOUT` setting is ignored).
|
||||
|
||||
|
||||
.. _faq-specific-reactor:
|
||||
|
||||
How to deal with ``<class 'ValueError'>: filedescriptor out of range in select()`` exceptions?
|
||||
----------------------------------------------------------------------------------------------
|
||||
|
||||
This issue `has been reported`_ to appear when running broad crawls in macOS, where the default
|
||||
Twisted reactor is :class:`twisted.internet.selectreactor.SelectReactor`. Switching to a
|
||||
different reactor is possible by using the :setting:`TWISTED_REACTOR` setting.
|
||||
|
||||
|
||||
.. _has been reported: https://github.com/scrapy/scrapy/issues/2905
|
||||
.. _user agents: https://en.wikipedia.org/wiki/User_agent
|
||||
.. _LIFO: https://en.wikipedia.org/wiki/Stack_(abstract_data_type)
|
||||
.. _DFO order: https://en.wikipedia.org/wiki/Depth-first_search
|
||||
|
|
|
|||
|
|
@ -170,7 +170,7 @@ Solving specific problems
|
|||
Get answers to most frequently asked questions.
|
||||
|
||||
:doc:`topics/debug`
|
||||
Learn how to debug common problems of your scrapy spider.
|
||||
Learn how to debug common problems of your Scrapy spider.
|
||||
|
||||
:doc:`topics/contracts`
|
||||
Learn how to use contracts for testing your spiders.
|
||||
|
|
|
|||
|
|
@ -7,12 +7,12 @@ Installation guide
|
|||
Installing Scrapy
|
||||
=================
|
||||
|
||||
Scrapy runs on Python 2.7 and Python 3.5 or above
|
||||
Scrapy runs on Python 3.5 or above
|
||||
under CPython (default Python implementation) and PyPy (starting with PyPy 5.9).
|
||||
|
||||
If you're using `Anaconda`_ or `Miniconda`_, you can install the package from
|
||||
the `conda-forge`_ channel, which has up-to-date packages for Linux, Windows
|
||||
and OS X.
|
||||
and macOS.
|
||||
|
||||
To install Scrapy using ``conda``, run::
|
||||
|
||||
|
|
@ -65,7 +65,7 @@ please refer to their respective installation instructions:
|
|||
* `lxml installation`_
|
||||
* `cryptography installation`_
|
||||
|
||||
.. _lxml installation: http://lxml.de/installation.html
|
||||
.. _lxml installation: https://lxml.de/installation.html
|
||||
.. _cryptography installation: https://cryptography.io/en/latest/installation/
|
||||
|
||||
|
||||
|
|
@ -78,40 +78,21 @@ TL;DR: We recommend installing Scrapy inside a virtual environment
|
|||
on all platforms.
|
||||
|
||||
Python packages can be installed either globally (a.k.a system wide),
|
||||
or in user-space. We do not recommend installing scrapy system wide.
|
||||
or in user-space. We do not recommend installing Scrapy system wide.
|
||||
|
||||
Instead, we recommend that you install scrapy within a so-called
|
||||
"virtual environment" (`virtualenv`_).
|
||||
Virtualenvs allow you to not conflict with already-installed Python
|
||||
Instead, we recommend that you install Scrapy within a so-called
|
||||
"virtual environment" (:mod:`venv`).
|
||||
Virtual environments allow you to not conflict with already-installed Python
|
||||
system packages (which could break some of your system tools and scripts),
|
||||
and still install packages normally with ``pip`` (without ``sudo`` and the likes).
|
||||
|
||||
To get started with virtual environments, see `virtualenv installation instructions`_.
|
||||
To install it globally (having it globally installed actually helps here),
|
||||
it should be a matter of running::
|
||||
See :ref:`tut-venv` on how to create your virtual environment.
|
||||
|
||||
$ [sudo] pip install virtualenv
|
||||
|
||||
Check this `user guide`_ on how to create your virtualenv.
|
||||
|
||||
.. note::
|
||||
If you use Linux or OS X, `virtualenvwrapper`_ is a handy tool to create virtualenvs.
|
||||
|
||||
Once you have created a virtualenv, you can install scrapy inside it with ``pip``,
|
||||
Once you have created a virtual environment, you can install Scrapy inside it with ``pip``,
|
||||
just like any other Python package.
|
||||
(See :ref:`platform-specific guides <intro-install-platform-notes>`
|
||||
below for non-Python dependencies that you may need to install beforehand).
|
||||
|
||||
Python virtualenvs can be created to use Python 2 by default, or Python 3 by default.
|
||||
|
||||
* If you want to install scrapy with Python 3, install scrapy within a Python 3 virtualenv.
|
||||
* And if you want to install scrapy with Python 2, install scrapy within a Python 2 virtualenv.
|
||||
|
||||
.. _virtualenv: https://virtualenv.pypa.io
|
||||
.. _virtualenv installation instructions: https://virtualenv.pypa.io/en/stable/installation/
|
||||
.. _virtualenvwrapper: https://virtualenvwrapper.readthedocs.io/en/latest/install.html
|
||||
.. _user guide: https://virtualenv.pypa.io/en/stable/userguide/
|
||||
|
||||
|
||||
.. _intro-install-platform-notes:
|
||||
|
||||
|
|
@ -146,19 +127,15 @@ albeit with potential issues with TLS connections.
|
|||
typically too old and slow to catch up with latest Scrapy.
|
||||
|
||||
|
||||
To install scrapy on Ubuntu (or Ubuntu-based) systems, you need to install
|
||||
To install Scrapy on Ubuntu (or Ubuntu-based) systems, you need to install
|
||||
these dependencies::
|
||||
|
||||
sudo apt-get install python-dev python-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev
|
||||
sudo apt-get install python3 python3-dev python3-pip libxml2-dev libxslt1-dev zlib1g-dev libffi-dev libssl-dev
|
||||
|
||||
- ``python-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev``
|
||||
- ``python3-dev``, ``zlib1g-dev``, ``libxml2-dev`` and ``libxslt1-dev``
|
||||
are required for ``lxml``
|
||||
- ``libssl-dev`` and ``libffi-dev`` are required for ``cryptography``
|
||||
|
||||
If you want to install scrapy on Python 3, you’ll also need Python 3 development headers::
|
||||
|
||||
sudo apt-get install python3 python3-dev
|
||||
|
||||
Inside a :ref:`virtualenv <intro-using-virtualenv>`,
|
||||
you can install Scrapy with ``pip`` after that::
|
||||
|
||||
|
|
@ -171,11 +148,11 @@ you can install Scrapy with ``pip`` after that::
|
|||
|
||||
.. _intro-install-macos:
|
||||
|
||||
Mac OS X
|
||||
--------
|
||||
macOS
|
||||
-----
|
||||
|
||||
Building Scrapy's dependencies requires the presence of a C compiler and
|
||||
development headers. On OS X this is typically provided by Apple’s Xcode
|
||||
development headers. On macOS this is typically provided by Apple’s Xcode
|
||||
development tools. To install the Xcode command line tools open a terminal
|
||||
window and run::
|
||||
|
||||
|
|
@ -211,15 +188,12 @@ solutions:
|
|||
|
||||
brew update; brew upgrade python
|
||||
|
||||
* *(Optional)* Install Scrapy inside an isolated python environment.
|
||||
* *(Optional)* :ref:`Install Scrapy inside a Python virtual environment
|
||||
<intro-using-virtualenv>`.
|
||||
|
||||
This method is a workaround for the above OS X issue, but it's an overall
|
||||
This method is a workaround for the above macOS issue, but it's an overall
|
||||
good practice for managing dependencies and can complement the first method.
|
||||
|
||||
`virtualenv`_ is a tool you can use to create virtual environments in python.
|
||||
We recommended reading a tutorial like
|
||||
http://docs.python-guide.org/en/latest/dev/virtualenvs/ to get started.
|
||||
|
||||
After any of these workarounds you should be able to install Scrapy::
|
||||
|
||||
pip install Scrapy
|
||||
|
|
@ -231,17 +205,17 @@ PyPy
|
|||
We recommend using the latest PyPy version. The version tested is 5.9.0.
|
||||
For PyPy3, only Linux installation was tested.
|
||||
|
||||
Most scrapy dependencies now have binary wheels for CPython, but not for PyPy.
|
||||
Most Scrapy dependencies now have binary wheels for CPython, but not for PyPy.
|
||||
This means that these dependencies will be built during installation.
|
||||
On OS X, you are likely to face an issue with building Cryptography dependency. The
|
||||
On macOS, you are likely to face an issue with building the Cryptography dependency. The
|
||||
solution to this problem is described
|
||||
`here <https://github.com/pyca/cryptography/issues/2692#issuecomment-272773481>`_,
|
||||
that is to ``brew install openssl`` and then export the flags that this command
|
||||
recommends (only needed when installing scrapy). Installing on Linux has no special
|
||||
recommends (only needed when installing Scrapy). Installing on Linux has no special
|
||||
issues besides installing build dependencies.
|
||||
Installing scrapy with PyPy on Windows is not tested.
|
||||
Installing Scrapy with PyPy on Windows is not tested.
|
||||
|
||||
You can check that scrapy is installed correctly by running ``scrapy bench``.
|
||||
You can check that Scrapy is installed correctly by running ``scrapy bench``.
|
||||
If this command gives errors such as
|
||||
``TypeError: ... got 2 unexpected keyword arguments``, this means
|
||||
that setuptools was unable to pick up one PyPy-specific dependency.
|
||||
|
|
@ -278,12 +252,12 @@ For details, see `Issue #2473 <https://github.com/scrapy/scrapy/issues/2473>`_.
|
|||
|
||||
.. _Python: https://www.python.org/
|
||||
.. _pip: https://pip.pypa.io/en/latest/installing/
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _parsel: https://pypi.python.org/pypi/parsel
|
||||
.. _w3lib: https://pypi.python.org/pypi/w3lib
|
||||
.. _twisted: https://twistedmatrix.com/
|
||||
.. _cryptography: https://cryptography.io/
|
||||
.. _pyOpenSSL: https://pypi.python.org/pypi/pyOpenSSL
|
||||
.. _lxml: https://lxml.de/index.html
|
||||
.. _parsel: https://pypi.org/project/parsel/
|
||||
.. _w3lib: https://pypi.org/project/w3lib/
|
||||
.. _twisted: https://twistedmatrix.com/trac/
|
||||
.. _cryptography: https://cryptography.io/en/latest/
|
||||
.. _pyOpenSSL: https://pypi.org/project/pyOpenSSL/
|
||||
.. _setuptools: https://pypi.python.org/pypi/setuptools
|
||||
.. _AUR Scrapy package: https://aur.archlinux.org/packages/scrapy/
|
||||
.. _homebrew: https://brew.sh/
|
||||
|
|
|
|||
|
|
@ -34,8 +34,8 @@ http://quotes.toscrape.com, following the pagination::
|
|||
def parse(self, response):
|
||||
for quote in response.css('div.quote'):
|
||||
yield {
|
||||
'text': quote.css('span.text::text').get(),
|
||||
'author': quote.xpath('span/small/text()').get(),
|
||||
'text': quote.css('span.text::text').get(),
|
||||
}
|
||||
|
||||
next_page = response.css('li.next a::attr("href")').get()
|
||||
|
|
|
|||
|
|
@ -212,7 +212,7 @@ using the :ref:`Scrapy shell <topics-shell>`. Run::
|
|||
.. note::
|
||||
|
||||
Remember to always enclose urls in quotes when running Scrapy shell from
|
||||
command-line, otherwise urls containing arguments (ie. ``&`` character)
|
||||
command-line, otherwise urls containing arguments (i.e. ``&`` character)
|
||||
will not work.
|
||||
|
||||
On Windows, use double quotes instead::
|
||||
|
|
@ -235,13 +235,16 @@ You will see something like::
|
|||
[s] shelp() Shell help (print this help)
|
||||
[s] fetch(req_or_url) Fetch request (or URL) and update local objects
|
||||
[s] view(response) View response in a browser
|
||||
>>>
|
||||
|
||||
Using the shell, you can try selecting elements using `CSS`_ with the response
|
||||
object::
|
||||
object:
|
||||
|
||||
>>> response.css('title')
|
||||
[<Selector xpath='descendant-or-self::title' data='<title>Quotes to Scrape</title>'>]
|
||||
.. invisible-code-block: python
|
||||
|
||||
response = load_response('http://quotes.toscrape.com/page/1/', 'quotes1.html')
|
||||
|
||||
>>> response.css('title')
|
||||
[<Selector xpath='descendant-or-self::title' data='<title>Quotes to Scrape</title>'>]
|
||||
|
||||
The result of running ``response.css('title')`` is a list-like object called
|
||||
:class:`~scrapy.selector.SelectorList`, which represents a list of
|
||||
|
|
@ -249,30 +252,30 @@ The result of running ``response.css('title')`` is a list-like object called
|
|||
and allow you to run further queries to fine-grain the selection or extract the
|
||||
data.
|
||||
|
||||
To extract the text from the title above, you can do::
|
||||
To extract the text from the title above, you can do:
|
||||
|
||||
>>> response.css('title::text').getall()
|
||||
['Quotes to Scrape']
|
||||
>>> response.css('title::text').getall()
|
||||
['Quotes to Scrape']
|
||||
|
||||
There are two things to note here: one is that we've added ``::text`` to the
|
||||
CSS query, to mean we want to select only the text elements directly inside
|
||||
``<title>`` element. If we don't specify ``::text``, we'd get the full title
|
||||
element, including its tags::
|
||||
element, including its tags:
|
||||
|
||||
>>> response.css('title').getall()
|
||||
['<title>Quotes to Scrape</title>']
|
||||
>>> response.css('title').getall()
|
||||
['<title>Quotes to Scrape</title>']
|
||||
|
||||
The other thing is that the result of calling ``.getall()`` is a list: it is
|
||||
possible that a selector returns more than one result, so we extract them all.
|
||||
When you know you just want the first result, as in this case, you can do::
|
||||
When you know you just want the first result, as in this case, you can do:
|
||||
|
||||
>>> response.css('title::text').get()
|
||||
'Quotes to Scrape'
|
||||
>>> response.css('title::text').get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
As an alternative, you could've written::
|
||||
As an alternative, you could've written:
|
||||
|
||||
>>> response.css('title::text')[0].get()
|
||||
'Quotes to Scrape'
|
||||
>>> response.css('title::text')[0].get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
However, using ``.get()`` directly on a :class:`~scrapy.selector.SelectorList`
|
||||
instance avoids an ``IndexError`` and returns ``None`` when it doesn't
|
||||
|
|
@ -285,14 +288,14 @@ to be scraped, you can at least get **some** data.
|
|||
Besides the :meth:`~scrapy.selector.SelectorList.getall` and
|
||||
:meth:`~scrapy.selector.SelectorList.get` methods, you can also use
|
||||
the :meth:`~scrapy.selector.SelectorList.re` method to extract using `regular
|
||||
expressions`_::
|
||||
expressions`_:
|
||||
|
||||
>>> response.css('title::text').re(r'Quotes.*')
|
||||
['Quotes to Scrape']
|
||||
>>> response.css('title::text').re(r'Q\w+')
|
||||
['Quotes']
|
||||
>>> response.css('title::text').re(r'(\w+) to (\w+)')
|
||||
['Quotes', 'Scrape']
|
||||
>>> response.css('title::text').re(r'Quotes.*')
|
||||
['Quotes to Scrape']
|
||||
>>> response.css('title::text').re(r'Q\w+')
|
||||
['Quotes']
|
||||
>>> response.css('title::text').re(r'(\w+) to (\w+)')
|
||||
['Quotes', 'Scrape']
|
||||
|
||||
In order to find the proper CSS selectors to use, you might find useful opening
|
||||
the response page from the shell in your web browser using ``view(response)``.
|
||||
|
|
@ -303,18 +306,18 @@ with a selector (see :ref:`topics-developer-tools`).
|
|||
visually selected elements, which works in many browsers.
|
||||
|
||||
.. _regular expressions: https://docs.python.org/3/library/re.html
|
||||
.. _Selector Gadget: http://selectorgadget.com/
|
||||
.. _Selector Gadget: https://selectorgadget.com/
|
||||
|
||||
|
||||
XPath: a brief intro
|
||||
^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions::
|
||||
Besides `CSS`_, Scrapy selectors also support using `XPath`_ expressions:
|
||||
|
||||
>>> response.xpath('//title')
|
||||
[<Selector xpath='//title' data='<title>Quotes to Scrape</title>'>]
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Quotes to Scrape'
|
||||
>>> response.xpath('//title')
|
||||
[<Selector xpath='//title' data='<title>Quotes to Scrape</title>'>]
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Quotes to Scrape'
|
||||
|
||||
XPath expressions are very powerful, and are the foundation of Scrapy
|
||||
Selectors. In fact, CSS selectors are converted to XPath under-the-hood. You
|
||||
|
|
@ -334,7 +337,7 @@ recommend `this tutorial to learn XPath through examples
|
|||
<http://zvon.org/comp/r/tut-XPath_1.html>`_, and `this tutorial to learn "how
|
||||
to think in XPath" <http://plasmasturm.org/log/xpath101/>`_.
|
||||
|
||||
.. _XPath: https://www.w3.org/TR/xpath
|
||||
.. _XPath: https://www.w3.org/TR/xpath/all/
|
||||
.. _CSS: https://www.w3.org/TR/selectors
|
||||
|
||||
Extracting quotes and authors
|
||||
|
|
@ -369,45 +372,53 @@ we want::
|
|||
|
||||
$ scrapy shell 'http://quotes.toscrape.com'
|
||||
|
||||
We get a list of selectors for the quote HTML elements with::
|
||||
We get a list of selectors for the quote HTML elements with:
|
||||
|
||||
>>> response.css("div.quote")
|
||||
>>> response.css("div.quote")
|
||||
[<Selector xpath="descendant-or-self::div[@class and contains(concat(' ', normalize-space(@class), ' '), ' quote ')]" data='<div class="quote" itemscope itemtype...'>,
|
||||
<Selector xpath="descendant-or-self::div[@class and contains(concat(' ', normalize-space(@class), ' '), ' quote ')]" data='<div class="quote" itemscope itemtype...'>,
|
||||
...]
|
||||
|
||||
Each of the selectors returned by the query above allows us to run further
|
||||
queries over their sub-elements. Let's assign the first selector to a
|
||||
variable, so that we can run our CSS selectors directly on a particular quote::
|
||||
variable, so that we can run our CSS selectors directly on a particular quote:
|
||||
|
||||
>>> quote = response.css("div.quote")[0]
|
||||
>>> quote = response.css("div.quote")[0]
|
||||
|
||||
Now, let's extract ``text``, ``author`` and the ``tags`` from that quote
|
||||
using the ``quote`` object we just created::
|
||||
using the ``quote`` object we just created:
|
||||
|
||||
>>> text = quote.css("span.text::text").get()
|
||||
>>> text
|
||||
'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'
|
||||
>>> author = quote.css("small.author::text").get()
|
||||
>>> author
|
||||
'Albert Einstein'
|
||||
>>> text = quote.css("span.text::text").get()
|
||||
>>> text
|
||||
'“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'
|
||||
>>> author = quote.css("small.author::text").get()
|
||||
>>> author
|
||||
'Albert Einstein'
|
||||
|
||||
Given that the tags are a list of strings, we can use the ``.getall()`` method
|
||||
to get all of them::
|
||||
to get all of them:
|
||||
|
||||
>>> tags = quote.css("div.tags a.tag::text").getall()
|
||||
>>> tags
|
||||
['change', 'deep-thoughts', 'thinking', 'world']
|
||||
>>> tags = quote.css("div.tags a.tag::text").getall()
|
||||
>>> tags
|
||||
['change', 'deep-thoughts', 'thinking', 'world']
|
||||
|
||||
.. invisible-code-block: python
|
||||
|
||||
from sys import version_info
|
||||
|
||||
.. skip: next if(version_info < (3, 6), reason="Only Python 3.6+ dictionaries match the output")
|
||||
|
||||
Having figured out how to extract each bit, we can now iterate over all the
|
||||
quotes elements and put them together into a Python dictionary::
|
||||
quotes elements and put them together into a Python dictionary:
|
||||
|
||||
>>> for quote in response.css("div.quote"):
|
||||
... text = quote.css("span.text::text").get()
|
||||
... author = quote.css("small.author::text").get()
|
||||
... tags = quote.css("div.tags a.tag::text").getall()
|
||||
... print(dict(text=text, author=author, tags=tags))
|
||||
{'tags': ['change', 'deep-thoughts', 'thinking', 'world'], 'author': 'Albert Einstein', 'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”'}
|
||||
{'tags': ['abilities', 'choices'], 'author': 'J.K. Rowling', 'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”'}
|
||||
... a few more of these, omitted for brevity
|
||||
>>>
|
||||
>>> for quote in response.css("div.quote"):
|
||||
... text = quote.css("span.text::text").get()
|
||||
... author = quote.css("small.author::text").get()
|
||||
... tags = quote.css("div.tags a.tag::text").getall()
|
||||
... print(dict(text=text, author=author, tags=tags))
|
||||
{'text': '“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”', 'author': 'Albert Einstein', 'tags': ['change', 'deep-thoughts', 'thinking', 'world']}
|
||||
{'text': '“It is our choices, Harry, that show what we truly are, far more than our abilities.”', 'author': 'J.K. Rowling', 'tags': ['abilities', 'choices']}
|
||||
...
|
||||
|
||||
Extracting data in our spider
|
||||
-----------------------------
|
||||
|
|
@ -505,23 +516,23 @@ markup:
|
|||
</li>
|
||||
</ul>
|
||||
|
||||
We can try extracting it in the shell::
|
||||
We can try extracting it in the shell:
|
||||
|
||||
>>> response.css('li.next a').get()
|
||||
'<a href="/page/2/">Next <span aria-hidden="true">→</span></a>'
|
||||
>>> response.css('li.next a').get()
|
||||
'<a href="/page/2/">Next <span aria-hidden="true">→</span></a>'
|
||||
|
||||
This gets the anchor element, but we want the attribute ``href``. For that,
|
||||
Scrapy supports a CSS extension that lets you select the attribute contents,
|
||||
like this::
|
||||
like this:
|
||||
|
||||
>>> response.css('li.next a::attr(href)').get()
|
||||
'/page/2/'
|
||||
>>> response.css('li.next a::attr(href)').get()
|
||||
'/page/2/'
|
||||
|
||||
There is also an ``attrib`` property available
|
||||
(see :ref:`selecting-attributes` for more)::
|
||||
(see :ref:`selecting-attributes` for more):
|
||||
|
||||
>>> response.css('li.next a').attrib['href']
|
||||
'/page/2'
|
||||
>>> response.css('li.next a').attrib['href']
|
||||
'/page/2/'
|
||||
|
||||
Let's see now our spider modified to recursively follow the link to the next
|
||||
page, extracting data from it::
|
||||
|
|
@ -605,21 +616,25 @@ instance; you still have to yield this Request.
|
|||
You can also pass a selector to ``response.follow`` instead of a string;
|
||||
this selector should extract necessary attributes::
|
||||
|
||||
for href in response.css('li.next a::attr(href)'):
|
||||
for href in response.css('ul.pager a::attr(href)'):
|
||||
yield response.follow(href, callback=self.parse)
|
||||
|
||||
For ``<a>`` elements there is a shortcut: ``response.follow`` uses their href
|
||||
attribute automatically. So the code can be shortened further::
|
||||
|
||||
for a in response.css('li.next a'):
|
||||
for a in response.css('ul.pager a'):
|
||||
yield response.follow(a, callback=self.parse)
|
||||
|
||||
.. note::
|
||||
To create multiple requests from an iterable, you can use
|
||||
:meth:`response.follow_all <scrapy.http.TextResponse.follow_all>` instead::
|
||||
|
||||
anchors = response.css('ul.pager a')
|
||||
yield from response.follow_all(anchors, callback=self.parse)
|
||||
|
||||
or, shortening it further::
|
||||
|
||||
yield from response.follow_all(css='ul.pager a', callback=self.parse)
|
||||
|
||||
``response.follow(response.css('li.next a'))`` is not valid because
|
||||
``response.css`` returns a list-like object with selectors for all results,
|
||||
not a single selector. A ``for`` loop like in the example above, or
|
||||
``response.follow(response.css('li.next a')[0])`` is fine.
|
||||
|
||||
More examples and patterns
|
||||
--------------------------
|
||||
|
|
@ -636,13 +651,11 @@ this time for scraping author information::
|
|||
start_urls = ['http://quotes.toscrape.com/']
|
||||
|
||||
def parse(self, response):
|
||||
# follow links to author pages
|
||||
for href in response.css('.author + a::attr(href)'):
|
||||
yield response.follow(href, self.parse_author)
|
||||
author_page_links = response.css('.author + a')
|
||||
yield from response.follow_all(author_page_links, self.parse_author)
|
||||
|
||||
# follow pagination links
|
||||
for href in response.css('li.next a::attr(href)'):
|
||||
yield response.follow(href, self.parse)
|
||||
pagination_links = response.css('li.next a')
|
||||
yield from response.follow_all(pagination_links, self.parse)
|
||||
|
||||
def parse_author(self, response):
|
||||
def extract_with_css(query):
|
||||
|
|
@ -658,8 +671,10 @@ This spider will start from the main page, it will follow all the links to the
|
|||
authors pages calling the ``parse_author`` callback for each of them, and also
|
||||
the pagination links with the ``parse`` callback as we saw before.
|
||||
|
||||
Here we're passing callbacks to ``response.follow`` as positional arguments
|
||||
to make the code shorter; it also works for ``scrapy.Request``.
|
||||
Here we're passing callbacks to
|
||||
:meth:`response.follow_all <scrapy.http.TextResponse.follow_all>` as positional
|
||||
arguments to make the code shorter; it also works for
|
||||
:class:`~scrapy.http.Request`.
|
||||
|
||||
The ``parse_author`` callback defines a helper function to extract and cleanup the
|
||||
data from a CSS query and yields the Python dict with the author data.
|
||||
|
|
|
|||
145
docs/news.rst
145
docs/news.rst
|
|
@ -26,7 +26,7 @@ Backward-incompatible changes
|
|||
* Python 3.4 is no longer supported, and some of the minimum requirements of
|
||||
Scrapy have also changed:
|
||||
|
||||
* cssselect_ 0.9.1
|
||||
* :doc:`cssselect <cssselect:index>` 0.9.1
|
||||
* cryptography_ 2.0
|
||||
* lxml_ 3.5.0
|
||||
* pyOpenSSL_ 16.2.0
|
||||
|
|
@ -47,13 +47,13 @@ Backward-incompatible changes
|
|||
(:issue:`2111`, :issue:`3392`, :issue:`3442`, :issue:`3450`)
|
||||
|
||||
* :class:`~scrapy.loader.ItemLoader` now turns the values of its input item
|
||||
into lists::
|
||||
into lists:
|
||||
|
||||
>>> item = MyItem()
|
||||
>>> item['field'] = 'value1'
|
||||
>>> loader = ItemLoader(item=item)
|
||||
>>> item['field']
|
||||
['value1']
|
||||
>>> item = MyItem()
|
||||
>>> item['field'] = 'value1'
|
||||
>>> loader = ItemLoader(item=item)
|
||||
>>> item['field']
|
||||
['value1']
|
||||
|
||||
This is needed to allow adding values to existing fields
|
||||
(``loader.add_value('field', 'value2')``).
|
||||
|
|
@ -288,6 +288,13 @@ Backward-incompatible changes
|
|||
:class:`~scrapy.http.Request` objects instead of arbitrary Python data
|
||||
structures.
|
||||
|
||||
* An additional ``crawler`` parameter has been added to the ``__init__`` method
|
||||
of the :class:`scrapy.core.scheduler.Scheduler` class.
|
||||
Custom scheduler subclasses which don't accept arbitrary parameters in
|
||||
their ``__init__`` method might break because of this change.
|
||||
|
||||
For more information, refer to the documentation for the :setting:`SCHEDULER` setting.
|
||||
|
||||
See also :ref:`1.7-deprecation-removals` below.
|
||||
|
||||
|
||||
|
|
@ -308,12 +315,12 @@ New features
|
|||
convenient way to build JSON requests (:issue:`3504`, :issue:`3505`)
|
||||
|
||||
* A ``process_request`` callback passed to the :class:`~scrapy.spiders.Rule`
|
||||
constructor now receives the :class:`~scrapy.http.Response` object that
|
||||
``__init__`` method now receives the :class:`~scrapy.http.Response` object that
|
||||
originated the request as its second argument (:issue:`3682`)
|
||||
|
||||
* A new ``restrict_text`` parameter for the
|
||||
:attr:`LinkExtractor <scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor>`
|
||||
constructor allows filtering links by linking text (:issue:`3622`,
|
||||
``__init__`` method allows filtering links by linking text (:issue:`3622`,
|
||||
:issue:`3635`)
|
||||
|
||||
* A new :setting:`FEED_STORAGE_S3_ACL` setting allows defining a custom ACL
|
||||
|
|
@ -479,7 +486,7 @@ The following deprecated APIs have been removed (:issue:`3578`):
|
|||
|
||||
* From :class:`~scrapy.selector.Selector`:
|
||||
|
||||
* ``_root`` (both the constructor argument and the object property, use
|
||||
* ``_root`` (both the ``__init__`` method argument and the object property, use
|
||||
``root``)
|
||||
|
||||
* ``extract_unquoted`` (use ``getall``)
|
||||
|
|
@ -678,7 +685,7 @@ Usability improvements
|
|||
* a message is added to IgnoreRequest in RobotsTxtMiddleware (:issue:`3113`)
|
||||
* better validation of ``url`` argument in ``Response.follow`` (:issue:`3131`)
|
||||
* non-zero exit code is returned from Scrapy commands when error happens
|
||||
on spider inititalization (:issue:`3226`)
|
||||
on spider initialization (:issue:`3226`)
|
||||
* Link extraction improvements: "ftp" is added to scheme list (:issue:`3152`);
|
||||
"flv" is added to common video extensions (:issue:`3165`)
|
||||
* better error message when an exporter is disabled (:issue:`3358`);
|
||||
|
|
@ -1069,7 +1076,7 @@ Cleanups & Refactoring
|
|||
~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
- Tests: remove temp files and folders (:issue:`2570`),
|
||||
fixed ProjectUtilsTest on OS X (:issue:`2569`),
|
||||
fixed ProjectUtilsTest on macOS (:issue:`2569`),
|
||||
use portable pypy for Linux on Travis CI (:issue:`2710`)
|
||||
- Separate building request from ``_requests_to_follow`` in CrawlSpider (:issue:`2562`)
|
||||
- Remove “Python 3 progress” badge (:issue:`2567`)
|
||||
|
|
@ -1156,7 +1163,7 @@ Bug fixes
|
|||
- Fix :command:`view` command ; it was a regression in v1.3.0 (:issue:`2503`).
|
||||
- Fix tests regarding ``*_EXPIRES settings`` with Files/Images pipelines (:issue:`2460`).
|
||||
- Fix name of generated pipeline class when using basic project template (:issue:`2466`).
|
||||
- Fix compatiblity with Twisted 17+ (:issue:`2496`, :issue:`2528`).
|
||||
- Fix compatibility with Twisted 17+ (:issue:`2496`, :issue:`2528`).
|
||||
- Fix ``scrapy.Item`` inheritance on Python 3.6 (:issue:`2511`).
|
||||
- Enforce numeric values for components order in ``SPIDER_MIDDLEWARES``,
|
||||
``DOWNLOADER_MIDDLEWARES``, ``EXTENIONS`` and ``SPIDER_CONTRACTS`` (:issue:`2420`).
|
||||
|
|
@ -1164,7 +1171,7 @@ Bug fixes
|
|||
Documentation
|
||||
~~~~~~~~~~~~~
|
||||
|
||||
- Reword Code of Coduct section and upgrade to Contributor Covenant v1.4
|
||||
- Reword Code of Conduct section and upgrade to Contributor Covenant v1.4
|
||||
(:issue:`2469`).
|
||||
- Clarify that passing spider arguments converts them to spider attributes
|
||||
(:issue:`2483`).
|
||||
|
|
@ -1178,7 +1185,7 @@ Documentation
|
|||
Cleanups
|
||||
~~~~~~~~
|
||||
|
||||
- Remove reduntant check in ``MetaRefreshMiddleware`` (:issue:`2542`).
|
||||
- Remove redundant check in ``MetaRefreshMiddleware`` (:issue:`2542`).
|
||||
- Faster checks in ``LinkExtractor`` for allow/deny patterns (:issue:`2538`).
|
||||
- Remove dead code supporting old Twisted versions (:issue:`2544`).
|
||||
|
||||
|
|
@ -1204,7 +1211,7 @@ New Features
|
|||
- ``MailSender`` now accepts single strings as values for ``to`` and ``cc``
|
||||
arguments (:issue:`2272`)
|
||||
- ``scrapy fetch url``, ``scrapy shell url`` and ``fetch(url)`` inside
|
||||
scrapy shell now follow HTTP redirections by default (:issue:`2290`);
|
||||
Scrapy shell now follow HTTP redirections by default (:issue:`2290`);
|
||||
See :command:`fetch` and :command:`shell` for details.
|
||||
- ``HttpErrorMiddleware`` now logs errors with ``INFO`` level instead of ``DEBUG``;
|
||||
this is technically **backward incompatible** so please check your log parsers.
|
||||
|
|
@ -1360,7 +1367,7 @@ Documentation
|
|||
|
||||
- Grammar fixes: :issue:`2128`, :issue:`1566`.
|
||||
- Download stats badge removed from README (:issue:`2160`).
|
||||
- New scrapy :ref:`architecture diagram <topics-architecture>` (:issue:`2165`).
|
||||
- New Scrapy :ref:`architecture diagram <topics-architecture>` (:issue:`2165`).
|
||||
- Updated ``Response`` parameters documentation (:issue:`2197`).
|
||||
- Reworded misleading :setting:`RANDOMIZE_DOWNLOAD_DELAY` description (:issue:`2190`).
|
||||
- Add StackOverflow as a support channel (:issue:`2257`).
|
||||
|
|
@ -1450,7 +1457,7 @@ Documentation
|
|||
- Use "url" variable in downloader middleware example (:issue:`2015`)
|
||||
- Grammar fixes (:issue:`2054`, :issue:`2120`)
|
||||
- New FAQ entry on using BeautifulSoup in spider callbacks (:issue:`2048`)
|
||||
- Add notes about scrapy not working on Windows with Python 3 (:issue:`2060`)
|
||||
- Add notes about Scrapy not working on Windows with Python 3 (:issue:`2060`)
|
||||
- Encourage complete titles in pull requests (:issue:`2026`)
|
||||
|
||||
Tests
|
||||
|
|
@ -1509,7 +1516,7 @@ This 1.1 release brings a lot of interesting features and bug fixes:
|
|||
You can use :setting:`FILES_STORE_S3_ACL` to change it.
|
||||
- We've reimplemented ``canonicalize_url()`` for more correct output,
|
||||
especially for URLs with non-ASCII characters (:issue:`1947`).
|
||||
This could change link extractors output compared to previous scrapy versions.
|
||||
This could change link extractors output compared to previous Scrapy versions.
|
||||
This may also invalidate some cache entries you could still have from pre-1.1 runs.
|
||||
**Warning: backward incompatible!**.
|
||||
|
||||
|
|
@ -1609,7 +1616,7 @@ Deprecations and Removals
|
|||
+ ``scrapy.utils.datatypes.SiteNode``
|
||||
|
||||
- The previously bundled ``scrapy.xlib.pydispatch`` library was deprecated and
|
||||
replaced by `pydispatcher <https://pypi.python.org/pypi/PyDispatcher>`_.
|
||||
replaced by `pydispatcher <https://pypi.org/project/PyDispatcher/>`_.
|
||||
|
||||
|
||||
Relocations
|
||||
|
|
@ -1638,7 +1645,7 @@ Bugfixes
|
|||
- Makes ``_monkeypatches`` more robust (:issue:`1634`).
|
||||
- Fixed bug on ``XMLItemExporter`` with non-string fields in
|
||||
items (:issue:`1738`).
|
||||
- Fixed startproject command in OS X (:issue:`1635`).
|
||||
- Fixed startproject command in macOS (:issue:`1635`).
|
||||
- Fixed :class:`~scrapy.exporters.PythonItemExporter` and CSVExporter for
|
||||
non-string item types (:issue:`1737`).
|
||||
- Various logging related fixes (:issue:`1294`, :issue:`1419`, :issue:`1263`,
|
||||
|
|
@ -1705,13 +1712,13 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
- fix ValueError: Invalid XPath: //div/[id="not-exists"]/text() on selectors.rst (:commit:`ca8d60f`)
|
||||
- Typos corrections (:commit:`7067117`)
|
||||
- fix typos in downloader-middleware.rst and exceptions.rst, middlware -> middleware (:commit:`32f115c`)
|
||||
- Add note to ubuntu install section about debian compatibility (:commit:`23fda69`)
|
||||
- Replace alternative OSX install workaround with virtualenv (:commit:`98b63ee`)
|
||||
- Add note to Ubuntu install section about Debian compatibility (:commit:`23fda69`)
|
||||
- Replace alternative macOS install workaround with virtualenv (:commit:`98b63ee`)
|
||||
- Reference Homebrew's homepage for installation instructions (:commit:`1925db1`)
|
||||
- Add oldest supported tox version to contributing docs (:commit:`5d10d6d`)
|
||||
- Note in install docs about pip being already included in python>=2.7.9 (:commit:`85c980e`)
|
||||
- Add non-python dependencies to Ubuntu install section in the docs (:commit:`fbd010d`)
|
||||
- Add OS X installation section to docs (:commit:`d8f4cba`)
|
||||
- Add macOS installation section to docs (:commit:`d8f4cba`)
|
||||
- DOC(ENH): specify path to rtd theme explicitly (:commit:`de73b1a`)
|
||||
- minor: scrapy.Spider docs grammar (:commit:`1ddcc7b`)
|
||||
- Make common practices sample code match the comments (:commit:`1b85bcf`)
|
||||
|
|
@ -1722,7 +1729,7 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
- Merge pull request #1513 from mgedmin/patch-2 (:commit:`5d4daf8`)
|
||||
- Typo (:commit:`f8d0682`)
|
||||
- Fix list formatting (:commit:`5f83a93`)
|
||||
- fix scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`)
|
||||
- fix Scrapy squeue tests after recent changes to queuelib (:commit:`3365c01`)
|
||||
- Merge pull request #1475 from rweindl/patch-1 (:commit:`2d688cd`)
|
||||
- Update tutorial.rst (:commit:`fbc1f25`)
|
||||
- Merge pull request #1449 from rhoekman/patch-1 (:commit:`7d6538c`)
|
||||
|
|
@ -1734,7 +1741,7 @@ Scrapy 1.0.4 (2015-12-30)
|
|||
Scrapy 1.0.3 (2015-08-11)
|
||||
-------------------------
|
||||
|
||||
- add service_identity to scrapy install_requires (:commit:`cbc2501`)
|
||||
- add service_identity to Scrapy install_requires (:commit:`cbc2501`)
|
||||
- Workaround for travis#296 (:commit:`66af9cd`)
|
||||
|
||||
.. _release-1.0.2:
|
||||
|
|
@ -1758,7 +1765,7 @@ Scrapy 1.0.1 (2015-07-01)
|
|||
- include tests/ to source distribution in MANIFEST.in (:commit:`eca227e`)
|
||||
- DOC Fix SelectJmes documentation (:commit:`b8567bc`)
|
||||
- DOC Bring Ubuntu and Archlinux outside of Windows subsection (:commit:`392233f`)
|
||||
- DOC remove version suffix from ubuntu package (:commit:`5303c66`)
|
||||
- DOC remove version suffix from Ubuntu package (:commit:`5303c66`)
|
||||
- DOC Update release date for 1.0 (:commit:`c89fa29`)
|
||||
|
||||
.. _release-1.0.0:
|
||||
|
|
@ -2211,7 +2218,7 @@ Scrapy 0.24.2 (2014-07-08)
|
|||
|
||||
- Use a mutable mapping to proxy deprecated settings.overrides and settings.defaults attribute (:commit:`e5e8133`)
|
||||
- there is not support for python3 yet (:commit:`3cd6146`)
|
||||
- Update python compatible version set to debian packages (:commit:`fa5d76b`)
|
||||
- Update python compatible version set to Debian packages (:commit:`fa5d76b`)
|
||||
- DOC fix formatting in release notes (:commit:`c6a9e20`)
|
||||
|
||||
Scrapy 0.24.1 (2014-06-27)
|
||||
|
|
@ -2229,12 +2236,12 @@ Enhancements
|
|||
|
||||
- Improve Scrapy top-level namespace (:issue:`494`, :issue:`684`)
|
||||
- Add selector shortcuts to responses (:issue:`554`, :issue:`690`)
|
||||
- Add new lxml based LinkExtractor to replace unmantained SgmlLinkExtractor
|
||||
- Add new lxml based LinkExtractor to replace unmaintained SgmlLinkExtractor
|
||||
(:issue:`559`, :issue:`761`, :issue:`763`)
|
||||
- Cleanup settings API - part of per-spider settings **GSoC project** (:issue:`737`)
|
||||
- Add UTF8 encoding header to templates (:issue:`688`, :issue:`762`)
|
||||
- Telnet console now binds to 127.0.0.1 by default (:issue:`699`)
|
||||
- Update debian/ubuntu install instructions (:issue:`509`, :issue:`549`)
|
||||
- Update Debian/Ubuntu install instructions (:issue:`509`, :issue:`549`)
|
||||
- Disable smart strings in lxml XPath evaluations (:issue:`535`)
|
||||
- Restore filesystem based cache as default for http
|
||||
cache middleware (:issue:`541`, :issue:`500`, :issue:`571`)
|
||||
|
|
@ -2267,7 +2274,7 @@ Enhancements
|
|||
- Tests and docs for ``request_fingerprint`` function (:issue:`597`)
|
||||
- Update SEP-19 for GSoC project ``per-spider settings`` (:issue:`705`)
|
||||
- Set exit code to non-zero when contracts fails (:issue:`727`)
|
||||
- Add a setting to control what class is instanciated as Downloader component
|
||||
- Add a setting to control what class is instantiated as Downloader component
|
||||
(:issue:`738`)
|
||||
- Pass response in ``item_dropped`` signal (:issue:`724`)
|
||||
- Improve ``scrapy check`` contracts command (:issue:`733`, :issue:`752`)
|
||||
|
|
@ -2276,7 +2283,7 @@ Enhancements
|
|||
- Add a note about reporting security issues (:issue:`697`)
|
||||
- Add LevelDB http cache storage backend (:issue:`626`, :issue:`500`)
|
||||
- Sort spider list output of ``scrapy list`` command (:issue:`742`)
|
||||
- Multiple documentation enhancemens and fixes
|
||||
- Multiple documentation enhancements and fixes
|
||||
(:issue:`575`, :issue:`587`, :issue:`590`, :issue:`596`, :issue:`610`,
|
||||
:issue:`617`, :issue:`618`, :issue:`627`, :issue:`613`, :issue:`643`,
|
||||
:issue:`654`, :issue:`675`, :issue:`663`, :issue:`711`, :issue:`714`)
|
||||
|
|
@ -2321,19 +2328,19 @@ Scrapy 0.22.1 (released 2014-02-08)
|
|||
- BaseSgmlLinkExtractor: Added unit test of a link with an inner tag (:commit:`c1cb418`)
|
||||
- BaseSgmlLinkExtractor: Fixed unknown_endtag() so that it only set current_link=None when the end tag match the opening tag (:commit:`7e4d627`)
|
||||
- Fix tests for Travis-CI build (:commit:`76c7e20`)
|
||||
- replace unencodeable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`)
|
||||
- replace unencodable codepoints with html entities. fixes #562 and #285 (:commit:`5f87b17`)
|
||||
- RegexLinkExtractor: encode URL unicode value when creating Links (:commit:`d0ee545`)
|
||||
- Updated the tutorial crawl output with latest output. (:commit:`8da65de`)
|
||||
- Updated shell docs with the crawler reference and fixed the actual shell output. (:commit:`875b9ab`)
|
||||
- PEP8 minor edits. (:commit:`f89efaf`)
|
||||
- Expose current crawler in the scrapy shell. (:commit:`5349cec`)
|
||||
- Expose current crawler in the Scrapy shell. (:commit:`5349cec`)
|
||||
- Unused re import and PEP8 minor edits. (:commit:`387f414`)
|
||||
- Ignore None's values when using the ItemLoader. (:commit:`0632546`)
|
||||
- DOC Fixed HTTPCACHE_STORAGE typo in the default value which is now Filesystem instead Dbm. (:commit:`cde9a8c`)
|
||||
- show ubuntu setup instructions as literal code (:commit:`fb5c9c5`)
|
||||
- show Ubuntu setup instructions as literal code (:commit:`fb5c9c5`)
|
||||
- Update Ubuntu installation instructions (:commit:`70fb105`)
|
||||
- Merge pull request #550 from stray-leone/patch-1 (:commit:`6f70b6a`)
|
||||
- modify the version of scrapy ubuntu package (:commit:`725900d`)
|
||||
- modify the version of Scrapy Ubuntu package (:commit:`725900d`)
|
||||
- fix 0.22.0 release date (:commit:`af0219a`)
|
||||
- fix typos in news.rst and remove (not released yet) header (:commit:`b7f58f4`)
|
||||
|
||||
|
|
@ -2354,7 +2361,7 @@ Enhancements
|
|||
- Improve test coverage and forthcoming Python 3 support (:issue:`525`)
|
||||
- Promote startup info on settings and middleware to INFO level (:issue:`520`)
|
||||
- Support partials in ``get_func_args`` util (:issue:`506`, issue:`504`)
|
||||
- Allow running indiviual tests via tox (:issue:`503`)
|
||||
- Allow running individual tests via tox (:issue:`503`)
|
||||
- Update extensions ignored by link extractors (:issue:`498`)
|
||||
- Add middleware methods to get files/images/thumbs paths (:issue:`490`)
|
||||
- Improve offsite middleware tests (:issue:`478`)
|
||||
|
|
@ -2411,7 +2418,7 @@ Enhancements
|
|||
- scrapy.mail.MailSender now can connect over TLS or upgrade using STARTTLS (:issue:`327`)
|
||||
- New FilesPipeline with functionality factored out from ImagesPipeline (:issue:`370`, :issue:`409`)
|
||||
- Recommend Pillow instead of PIL for image handling (:issue:`317`)
|
||||
- Added debian packages for Ubuntu quantal and raring (:commit:`86230c0`)
|
||||
- Added Debian packages for Ubuntu Quantal and Raring (:commit:`86230c0`)
|
||||
- Mock server (used for tests) can listen for HTTPS requests (:issue:`410`)
|
||||
- Remove multi spider support from multiple core components
|
||||
(:issue:`422`, :issue:`421`, :issue:`420`, :issue:`419`, :issue:`423`, :issue:`418`)
|
||||
|
|
@ -2430,7 +2437,7 @@ Bugfixes
|
|||
- Fix tests under Django 1.6 (:commit:`b6bed44c`)
|
||||
- Lot of bugfixes to retry middleware under disconnections using HTTP 1.1 download handler
|
||||
- Fix inconsistencies among Twisted releases (:issue:`406`)
|
||||
- Fix scrapy shell bugs (:issue:`418`, :issue:`407`)
|
||||
- Fix Scrapy shell bugs (:issue:`418`, :issue:`407`)
|
||||
- Fix invalid variable name in setup.py (:issue:`429`)
|
||||
- Fix tutorial references (:issue:`387`)
|
||||
- Improve request-response docs (:issue:`391`)
|
||||
|
|
@ -2443,7 +2450,7 @@ Other
|
|||
~~~~~
|
||||
|
||||
- Dropped Python 2.6 support (:issue:`448`)
|
||||
- Add `cssselect`_ python package as install dependency
|
||||
- Add :doc:`cssselect <cssselect:index>` python package as install dependency
|
||||
- Drop libxml2 and multi selector's backend support, `lxml`_ is required from now on.
|
||||
- Minimum Twisted version increased to 10.0.0, dropped Twisted 8.0 support.
|
||||
- Running test suite now requires ``mock`` python library (:issue:`390`)
|
||||
|
|
@ -2512,15 +2519,15 @@ Scrapy 0.18.1 (released 2013-08-27)
|
|||
- test PotentiaDataLoss errors on unbound responses (:commit:`b15470d`)
|
||||
- Treat responses without content-length or Transfer-Encoding as good responses (:commit:`c4bf324`)
|
||||
- do no include ResponseFailed if http11 handler is not enabled (:commit:`6cbe684`)
|
||||
- New HTTP client wraps connection losts in ResponseFailed exception. fix #373 (:commit:`1a20bba`)
|
||||
- New HTTP client wraps connection lost in ResponseFailed exception. fix #373 (:commit:`1a20bba`)
|
||||
- limit travis-ci build matrix (:commit:`3b01bb8`)
|
||||
- Merge pull request #375 from peterarenot/patch-1 (:commit:`fa766d7`)
|
||||
- Fixed so it refers to the correct folder (:commit:`3283809`)
|
||||
- added quantal & raring to support ubuntu releases (:commit:`1411923`)
|
||||
- added Quantal & Raring to support Ubuntu releases (:commit:`1411923`)
|
||||
- fix retry middleware which didn't retry certain connection errors after the upgrade to http1 client, closes GH-373 (:commit:`bb35ed0`)
|
||||
- fix XmlItemExporter in Python 2.7.4 and 2.7.5 (:commit:`de3e451`)
|
||||
- minor updates to 0.18 release notes (:commit:`c45e5f1`)
|
||||
- fix contributters list format (:commit:`0b60031`)
|
||||
- fix contributors list format (:commit:`0b60031`)
|
||||
|
||||
Scrapy 0.18.0 (released 2013-08-09)
|
||||
-----------------------------------
|
||||
|
|
@ -2555,8 +2562,8 @@ Scrapy 0.18.0 (released 2013-08-09)
|
|||
- Collect idle downloader slots (:issue:`297`)
|
||||
- Add ``ftp://`` scheme downloader handler (:issue:`329`)
|
||||
- Added downloader benchmark webserver and spider tools :ref:`benchmarking`
|
||||
- Moved persistent (on disk) queues to a separate project (queuelib_) which scrapy now depends on
|
||||
- Add scrapy commands using external libraries (:issue:`260`)
|
||||
- Moved persistent (on disk) queues to a separate project (queuelib_) which Scrapy now depends on
|
||||
- Add Scrapy commands using external libraries (:issue:`260`)
|
||||
- Added ``--pdb`` option to ``scrapy`` command line tool
|
||||
- Added :meth:`XPathSelector.remove_namespaces <scrapy.selector.Selector.remove_namespaces>` which allows to remove all namespaces from XML documents for convenience (to work with namespace-less XPaths). Documented in :ref:`topics-selectors`.
|
||||
- Several improvements to spider contracts
|
||||
|
|
@ -2564,11 +2571,11 @@ Scrapy 0.18.0 (released 2013-08-09)
|
|||
- MetaRefreshMiddldeware and RedirectMiddleware have different priorities to address #62
|
||||
- added from_crawler method to spiders
|
||||
- added system tests with mock server
|
||||
- more improvements to Mac OS compatibility (thanks Alex Cepoi)
|
||||
- more improvements to macOS compatibility (thanks Alex Cepoi)
|
||||
- several more cleanups to singletons and multi-spider support (thanks Nicolas Ramirez)
|
||||
- support custom download slots
|
||||
- added --spider option to "shell" command.
|
||||
- log overridden settings when scrapy starts
|
||||
- log overridden settings when Scrapy starts
|
||||
|
||||
Thanks to everyone who contribute to this release. Here is a list of
|
||||
contributors sorted by number of commits::
|
||||
|
|
@ -2617,7 +2624,7 @@ contributors sorted by number of commits::
|
|||
Scrapy 0.16.5 (released 2013-05-30)
|
||||
-----------------------------------
|
||||
|
||||
- obey request method when scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`)
|
||||
- obey request method when Scrapy deploy is redirected to a new endpoint (:commit:`8c4fcee`)
|
||||
- fix inaccurate downloader middleware documentation. refs #280 (:commit:`40667cb`)
|
||||
- doc: remove links to diveintopython.org, which is no longer available. closes #246 (:commit:`bd58bfa`)
|
||||
- Find form nodes in invalid html5 documents (:commit:`e3d6945`)
|
||||
|
|
@ -2631,8 +2638,8 @@ Scrapy 0.16.4 (released 2013-01-23)
|
|||
- Fixed error message formatting. log.err() doesn't support cool formatting and when error occurred, the message was: "ERROR: Error processing %(item)s" (:commit:`c16150c`)
|
||||
- lint and improve images pipeline error logging (:commit:`56b45fc`)
|
||||
- fixed doc typos (:commit:`243be84`)
|
||||
- add documentation topics: Broad Crawls & Common Practies (:commit:`1fbb715`)
|
||||
- fix bug in scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`)
|
||||
- add documentation topics: Broad Crawls & Common Practices (:commit:`1fbb715`)
|
||||
- fix bug in Scrapy parse command when spider is not specified explicitly. closes #209 (:commit:`c72e682`)
|
||||
- Update docs/topics/commands.rst (:commit:`28eac7a`)
|
||||
|
||||
Scrapy 0.16.3 (released 2012-12-07)
|
||||
|
|
@ -2640,7 +2647,7 @@ Scrapy 0.16.3 (released 2012-12-07)
|
|||
|
||||
- Remove concurrency limitation when using download delays and still ensure inter-request delays are enforced (:commit:`487b9b5`)
|
||||
- add error details when image pipeline fails (:commit:`8232569`)
|
||||
- improve mac os compatibility (:commit:`8dcf8aa`)
|
||||
- improve macOS compatibility (:commit:`8dcf8aa`)
|
||||
- setup.py: use README.rst to populate long_description (:commit:`7b5310d`)
|
||||
- doc: removed obsolete references to ClientForm (:commit:`80f9bb6`)
|
||||
- correct docs for default storage backend (:commit:`2aa491b`)
|
||||
|
|
@ -2651,11 +2658,11 @@ Scrapy 0.16.3 (released 2012-12-07)
|
|||
Scrapy 0.16.2 (released 2012-11-09)
|
||||
-----------------------------------
|
||||
|
||||
- scrapy contracts: python2.6 compat (:commit:`a4a9199`)
|
||||
- scrapy contracts verbose option (:commit:`ec41673`)
|
||||
- proper unittest-like output for scrapy contracts (:commit:`86635e4`)
|
||||
- Scrapy contracts: python2.6 compat (:commit:`a4a9199`)
|
||||
- Scrapy contracts verbose option (:commit:`ec41673`)
|
||||
- proper unittest-like output for Scrapy contracts (:commit:`86635e4`)
|
||||
- added open_in_browser to debugging doc (:commit:`c9b690d`)
|
||||
- removed reference to global scrapy stats from settings doc (:commit:`dd55067`)
|
||||
- removed reference to global Scrapy stats from settings doc (:commit:`dd55067`)
|
||||
- Fix SpiderState bug in Windows platforms (:commit:`58998f4`)
|
||||
|
||||
|
||||
|
|
@ -2665,7 +2672,7 @@ Scrapy 0.16.1 (released 2012-10-26)
|
|||
- fixed LogStats extension, which got broken after a wrong merge before the 0.16 release (:commit:`8c780fd`)
|
||||
- better backward compatibility for scrapy.conf.settings (:commit:`3403089`)
|
||||
- extended documentation on how to access crawler stats from extensions (:commit:`c4da0b5`)
|
||||
- removed .hgtags (no longer needed now that scrapy uses git) (:commit:`d52c188`)
|
||||
- removed .hgtags (no longer needed now that Scrapy uses git) (:commit:`d52c188`)
|
||||
- fix dashes under rst headers (:commit:`fa4f7f9`)
|
||||
- set release date for 0.16.0 in news (:commit:`e292246`)
|
||||
|
||||
|
|
@ -2680,8 +2687,7 @@ Scrapy changes:
|
|||
- documented :doc:`topics/autothrottle` and added to extensions installed by default. You still need to enable it with :setting:`AUTOTHROTTLE_ENABLED`
|
||||
- major Stats Collection refactoring: removed separation of global/per-spider stats, removed stats-related signals (``stats_spider_opened``, etc). Stats are much simpler now, backward compatibility is kept on the Stats Collector API and signals.
|
||||
- added :meth:`~scrapy.spidermiddlewares.SpiderMiddleware.process_start_requests` method to spider middlewares
|
||||
- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info.
|
||||
- dropped Signals singleton. Signals should now be accesed through the Crawler.signals attribute. See the signals documentation for more info.
|
||||
- dropped Signals singleton. Signals should now be accessed through the Crawler.signals attribute. See the signals documentation for more info.
|
||||
- dropped Stats Collector singleton. Stats can now be accessed through the Crawler.stats attribute. See the stats collection documentation for more info.
|
||||
- documented :ref:`topics-api`
|
||||
- ``lxml`` is now the default selectors backend instead of ``libxml2``
|
||||
|
|
@ -2703,7 +2709,7 @@ Scrapy changes:
|
|||
- removed ``ENCODING_ALIASES`` setting, as encoding auto-detection has been moved to the `w3lib`_ library
|
||||
- promoted :ref:`topics-djangoitem` to main contrib
|
||||
- LogFormatter method now return dicts(instead of strings) to support lazy formatting (:issue:`164`, :commit:`dcef7b0`)
|
||||
- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the constructor
|
||||
- downloader handlers (:setting:`DOWNLOAD_HANDLERS` setting) now receive settings as the first argument of the ``__init__`` method
|
||||
- replaced memory usage acounting with (more portable) `resource`_ module, removed ``scrapy.utils.memory`` module
|
||||
- removed signal: ``scrapy.mail.mail_sent``
|
||||
- removed ``TRACK_REFS`` setting, now :ref:`trackrefs <topics-leaks-trackrefs>` is always enabled
|
||||
|
|
@ -2715,7 +2721,7 @@ Scrapy changes:
|
|||
Scrapy 0.14.4
|
||||
-------------
|
||||
|
||||
- added precise to supported ubuntu distros (:commit:`b7e46df`)
|
||||
- added precise to supported Ubuntu distros (:commit:`b7e46df`)
|
||||
- fixed bug in json-rpc webservice reported in https://groups.google.com/forum/#!topic/scrapy-users/qgVBmFybNAQ/discussion. also removed no longer supported 'run' command from extras/scrapy-ws.py (:commit:`340fbdb`)
|
||||
- meta tag attributes for content-type http equiv can be in any order. #123 (:commit:`0cb68af`)
|
||||
- replace "import Image" by more standard "from PIL import Image". closes #88 (:commit:`4d17048`)
|
||||
|
|
@ -2728,11 +2734,11 @@ Scrapy 0.14.3
|
|||
- include egg files used by testsuite in source distribution. #118 (:commit:`c897793`)
|
||||
- update docstring in project template to avoid confusion with genspider command, which may be considered as an advanced feature. refs #107 (:commit:`2548dcc`)
|
||||
- added note to docs/topics/firebug.rst about google directory being shut down (:commit:`668e352`)
|
||||
- dont discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`)
|
||||
- don't discard slot when empty, just save in another dict in order to recycle if needed again. (:commit:`8e9f607`)
|
||||
- do not fail handling unicode xpaths in libxml2 backed selectors (:commit:`b830e95`)
|
||||
- fixed minor mistake in Request objects documentation (:commit:`bf3c9ee`)
|
||||
- fixed minor defect in link extractors documentation (:commit:`ba14f38`)
|
||||
- removed some obsolete remaining code related to sqlite support in scrapy (:commit:`0665175`)
|
||||
- removed some obsolete remaining code related to sqlite support in Scrapy (:commit:`0665175`)
|
||||
|
||||
Scrapy 0.14.2
|
||||
-------------
|
||||
|
|
@ -2917,7 +2923,7 @@ API changes
|
|||
- ``Request.copy()`` and ``Request.replace()`` now also copies their ``callback`` and ``errback`` attributes (#231)
|
||||
- Removed ``UrlFilterMiddleware`` from ``scrapy.contrib`` (already disabled by default)
|
||||
- Offsite middelware doesn't filter out any request coming from a spider that doesn't have a allowed_domains attribute (#225)
|
||||
- Removed Spider Manager ``load()`` method. Now spiders are loaded in the constructor itself.
|
||||
- Removed Spider Manager ``load()`` method. Now spiders are loaded in the ``__init__`` method itself.
|
||||
- Changes to Scrapy Manager (now called "Crawler"):
|
||||
- ``scrapy.core.manager.ScrapyManager`` class renamed to ``scrapy.crawler.Crawler``
|
||||
- ``scrapy.core.manager.scrapymanager`` singleton moved to ``scrapy.project.crawler``
|
||||
|
|
@ -3041,17 +3047,16 @@ Scrapy 0.7
|
|||
First release of Scrapy.
|
||||
|
||||
|
||||
.. _AJAX crawleable urls: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started?csw=1
|
||||
.. _AJAX crawleable urls: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started?csw=1
|
||||
.. _botocore: https://github.com/boto/botocore
|
||||
.. _chunked transfer encoding: https://en.wikipedia.org/wiki/Chunked_transfer_encoding
|
||||
.. _ClientForm: http://wwwsearch.sourceforge.net/old/ClientForm/
|
||||
.. _Creating a pull request: https://help.github.com/en/articles/creating-a-pull-request
|
||||
.. _cryptography: https://cryptography.io/en/latest/
|
||||
.. _cssselect: https://github.com/scrapy/cssselect/
|
||||
.. _docstrings: https://docs.python.org/glossary.html#term-docstring
|
||||
.. _KeyboardInterrupt: https://docs.python.org/library/exceptions.html#KeyboardInterrupt
|
||||
.. _docstrings: https://docs.python.org/3/glossary.html#term-docstring
|
||||
.. _KeyboardInterrupt: https://docs.python.org/3/library/exceptions.html#KeyboardInterrupt
|
||||
.. _LevelDB: https://github.com/google/leveldb
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _marshal: https://docs.python.org/2/library/marshal.html
|
||||
.. _parsel.csstranslator.GenericTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.GenericTranslator
|
||||
.. _parsel.csstranslator.HTMLTranslator: https://parsel.readthedocs.io/en/latest/parsel.html#parsel.csstranslator.HTMLTranslator
|
||||
|
|
@ -3062,11 +3067,11 @@ First release of Scrapy.
|
|||
.. _queuelib: https://github.com/scrapy/queuelib
|
||||
.. _registered with IANA: https://www.iana.org/assignments/media-types/media-types.xhtml
|
||||
.. _resource: https://docs.python.org/2/library/resource.html
|
||||
.. _robots.txt: http://www.robotstxt.org/
|
||||
.. _robots.txt: https://www.robotstxt.org/
|
||||
.. _scrapely: https://github.com/scrapy/scrapely
|
||||
.. _service_identity: https://service-identity.readthedocs.io/en/stable/
|
||||
.. _six: https://six.readthedocs.io/
|
||||
.. _tox: https://pypi.python.org/pypi/tox
|
||||
.. _tox: https://pypi.org/project/tox/
|
||||
.. _Twisted: https://twistedmatrix.com/trac/
|
||||
.. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/
|
||||
.. _w3lib: https://github.com/scrapy/w3lib
|
||||
|
|
|
|||
|
|
@ -1,2 +1,4 @@
|
|||
Sphinx>=2.1
|
||||
sphinx_rtd_theme
|
||||
sphinx-hoverxref
|
||||
sphinx-notfound-page
|
||||
sphinx_rtd_theme
|
||||
|
|
|
|||
|
|
@ -273,5 +273,3 @@ class (which they all inherit from).
|
|||
|
||||
Close the given spider. After this is called, no more specific stats
|
||||
can be accessed or collected.
|
||||
|
||||
.. _reactor: https://twistedmatrix.com/documents/current/core/howto/reactor-basics.html
|
||||
|
|
|
|||
|
|
@ -166,11 +166,10 @@ for concurrency.
|
|||
For more information about asynchronous programming and Twisted see these
|
||||
links:
|
||||
|
||||
* `Introduction to Deferreds in Twisted`_
|
||||
* :doc:`twisted:core/howto/defer-intro`
|
||||
* `Twisted - hello, asynchronous programming`_
|
||||
* `Twisted Introduction - Krondo`_
|
||||
|
||||
.. _Twisted: https://twistedmatrix.com/trac/
|
||||
.. _Introduction to Deferreds in Twisted: https://twistedmatrix.com/documents/current/core/howto/defer-intro.html
|
||||
.. _Twisted - hello, asynchronous programming: http://jessenoller.com/blog/2009/02/11/twisted-hello-asynchronous-programming/
|
||||
.. _Twisted Introduction - Krondo: http://krondo.com/an-introduction-to-asynchronous-programming-and-twisted/
|
||||
|
|
|
|||
|
|
@ -11,7 +11,7 @@ Design goals
|
|||
============
|
||||
|
||||
1. be nicer to sites instead of using default download delay of zero
|
||||
2. automatically adjust scrapy to the optimum crawling speed, so the user
|
||||
2. automatically adjust Scrapy to the optimum crawling speed, so the user
|
||||
doesn't have to tune the download delays to find the optimum one.
|
||||
The user only needs to specify the maximum concurrent requests
|
||||
it allows, and the extension does the rest.
|
||||
|
|
|
|||
|
|
@ -188,7 +188,7 @@ AjaxCrawlMiddleware helps to crawl them correctly.
|
|||
It is turned OFF by default because it has some performance overhead,
|
||||
and enabling it for focused crawls doesn't make much sense.
|
||||
|
||||
.. _ajax crawlable: https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
|
||||
.. _ajax crawlable: https://developers.google.com/search/docs/ajax-crawling/docs/getting-started
|
||||
|
||||
.. _broad-crawls-bfo:
|
||||
|
||||
|
|
@ -211,3 +211,10 @@ If your broad crawl shows a high memory usage, in addition to :ref:`crawling in
|
|||
BFO order <broad-crawls-bfo>` and :ref:`lowering concurrency
|
||||
<broad-crawls-concurrency>` you should :ref:`debug your memory leaks
|
||||
<topics-leaks>`.
|
||||
|
||||
|
||||
Install a specific Twisted reactor
|
||||
==================================
|
||||
|
||||
If the crawl is exceeding the system's capabilities, you might want to try
|
||||
installing a specific Twisted reactor, via the :setting:`TWISTED_REACTOR` setting.
|
||||
|
|
|
|||
|
|
@ -1,3 +1,5 @@
|
|||
.. highlight:: none
|
||||
|
||||
.. _topics-commands:
|
||||
|
||||
=================
|
||||
|
|
@ -27,7 +29,7 @@ in standard locations:
|
|||
1. ``/etc/scrapy.cfg`` or ``c:\scrapy\scrapy.cfg`` (system-wide),
|
||||
2. ``~/.config/scrapy.cfg`` (``$XDG_CONFIG_HOME``) and ``~/.scrapy.cfg`` (``$HOME``)
|
||||
for global (user-wide) settings, and
|
||||
3. ``scrapy.cfg`` inside a scrapy project's root (see next section).
|
||||
3. ``scrapy.cfg`` inside a Scrapy project's root (see next section).
|
||||
|
||||
Settings from these files are merged in the listed order of preference:
|
||||
user-defined values have higher priority than system-wide defaults
|
||||
|
|
@ -66,7 +68,9 @@ structure by default, similar to this::
|
|||
|
||||
The directory where the ``scrapy.cfg`` file resides is known as the *project
|
||||
root directory*. That file contains the name of the python module that defines
|
||||
the project settings. Here is an example::
|
||||
the project settings. Here is an example:
|
||||
|
||||
.. code-block:: ini
|
||||
|
||||
[settings]
|
||||
default = myproject.settings
|
||||
|
|
@ -80,7 +84,9 @@ A project root directory, the one that contains the ``scrapy.cfg``, may be
|
|||
shared by multiple Scrapy projects, each with its own settings module.
|
||||
|
||||
In that case, you must define one or more aliases for those settings modules
|
||||
under ``[settings]`` in your ``scrapy.cfg`` file::
|
||||
under ``[settings]`` in your ``scrapy.cfg`` file:
|
||||
|
||||
.. code-block:: ini
|
||||
|
||||
[settings]
|
||||
default = myproject1.settings
|
||||
|
|
@ -277,6 +283,8 @@ check
|
|||
|
||||
Run contract checks.
|
||||
|
||||
.. skip: start
|
||||
|
||||
Usage examples::
|
||||
|
||||
$ scrapy check -l
|
||||
|
|
@ -294,6 +302,8 @@ Usage examples::
|
|||
[FAILED] first_spider:parse
|
||||
>>> Returned 92 requests, expected 0..4
|
||||
|
||||
.. skip: end
|
||||
|
||||
.. command:: list
|
||||
|
||||
list
|
||||
|
|
@ -481,6 +491,8 @@ Supported options:
|
|||
|
||||
* ``--verbose`` or ``-v``: display information for each depth level
|
||||
|
||||
.. skip: start
|
||||
|
||||
Usage example::
|
||||
|
||||
$ scrapy parse http://www.example.com/ -c parse_item
|
||||
|
|
@ -495,6 +507,8 @@ Usage example::
|
|||
# Requests -----------------------------------------------------------------
|
||||
[]
|
||||
|
||||
.. skip: end
|
||||
|
||||
|
||||
.. command:: settings
|
||||
|
||||
|
|
@ -573,7 +587,9 @@ Default: ``''`` (empty string)
|
|||
A module to use for looking up custom Scrapy commands. This is used to add custom
|
||||
commands for your Scrapy project.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
COMMANDS_MODULE = 'mybot.commands'
|
||||
|
||||
|
|
@ -588,7 +604,11 @@ You can also add Scrapy commands from an external library by adding a
|
|||
``scrapy.commands`` section in the entry points of the library ``setup.py``
|
||||
file.
|
||||
|
||||
The following example adds ``my_command`` command::
|
||||
The following example adds ``my_command`` command:
|
||||
|
||||
.. skip: next
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
from setuptools import setup, find_packages
|
||||
|
||||
|
|
|
|||
|
|
@ -64,7 +64,7 @@ Use the :command:`check` command to run the contract checks.
|
|||
Custom Contracts
|
||||
================
|
||||
|
||||
If you find you need more power than the built-in scrapy contracts you can
|
||||
If you find you need more power than the built-in Scrapy contracts you can
|
||||
create and load your own contracts in the project by using the
|
||||
:setting:`SPIDER_CONTRACTS` setting::
|
||||
|
||||
|
|
|
|||
|
|
@ -5,7 +5,7 @@ Debugging Spiders
|
|||
=================
|
||||
|
||||
This document explains the most common techniques for debugging spiders.
|
||||
Consider the following scrapy spider below::
|
||||
Consider the following Scrapy spider below::
|
||||
|
||||
import scrapy
|
||||
from myproject.items import MyItem
|
||||
|
|
@ -48,6 +48,10 @@ The most basic way of checking the output of your spider is to use the
|
|||
of the spider at the method level. It has the advantage of being flexible and
|
||||
simple to use, but does not allow debugging code inside a method.
|
||||
|
||||
.. highlight:: none
|
||||
|
||||
.. skip: start
|
||||
|
||||
In order to see the item scraped from a specific url::
|
||||
|
||||
$ scrapy parse --spider=myspider -c parse_item -d 2 <item_url>
|
||||
|
|
@ -85,6 +89,8 @@ using::
|
|||
|
||||
$ scrapy parse --spider=myspider -d 3 'http://example.com/page1'
|
||||
|
||||
.. skip: end
|
||||
|
||||
|
||||
Scrapy Shell
|
||||
============
|
||||
|
|
@ -94,6 +100,8 @@ spider, it is of little help to check what happens inside a callback, besides
|
|||
showing the response received and the output. How to debug the situation when
|
||||
``parse_details`` sometimes receives no item?
|
||||
|
||||
.. highlight:: python
|
||||
|
||||
Fortunately, the :command:`shell` is your bread and butter in this case (see
|
||||
:ref:`topics-shell-inspect-response`)::
|
||||
|
||||
|
|
|
|||
|
|
@ -39,7 +39,7 @@ Therefore, you should keep in mind the following things:
|
|||
.. _topics-inspector:
|
||||
|
||||
Inspecting a website
|
||||
===================================
|
||||
====================
|
||||
|
||||
By far the most handy feature of the Developer Tools is the `Inspector`
|
||||
feature, which allows you to inspect the underlying HTML code of
|
||||
|
|
@ -79,13 +79,23 @@ sections and tags of a webpage, which greatly improves readability. You can
|
|||
expand and collapse a tag by clicking on the arrow in front of it or by double
|
||||
clicking directly on the tag. If we expand the ``span`` tag with the ``class=
|
||||
"text"`` we will see the quote-text we clicked on. The `Inspector` lets you
|
||||
copy XPaths to selected elements. Let's try it out: Right-click on the ``span``
|
||||
tag, select ``Copy > XPath`` and paste it in the scrapy shell like so::
|
||||
copy XPaths to selected elements. Let's try it out.
|
||||
|
||||
First open the Scrapy shell at http://quotes.toscrape.com/ in a terminal:
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
$ scrapy shell "http://quotes.toscrape.com/"
|
||||
(...)
|
||||
>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall()
|
||||
['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”]
|
||||
|
||||
Then, back to your web browser, right-click on the ``span`` tag, select
|
||||
``Copy > XPath`` and paste it in the Scrapy shell like so:
|
||||
|
||||
.. invisible-code-block: python
|
||||
|
||||
response = load_response('http://quotes.toscrape.com/', 'quotes.html')
|
||||
|
||||
>>> response.xpath('/html/body/div/div[2]/div[1]/div[1]/span[1]/text()').getall()
|
||||
['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”']
|
||||
|
||||
Adding ``text()`` at the end we are able to extract the first quote with this
|
||||
basic selector. But this XPath is not really that clever. All it does is
|
||||
|
|
@ -112,13 +122,13 @@ see each quote:
|
|||
|
||||
With this knowledge we can refine our XPath: Instead of a path to follow,
|
||||
we'll simply select all ``span`` tags with the ``class="text"`` by using
|
||||
the `has-class-extension`_::
|
||||
the `has-class-extension`_:
|
||||
|
||||
>>> response.xpath('//span[has-class("text")]/text()').getall()
|
||||
['"The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”,
|
||||
'“It is our choices, Harry, that show what we truly are, far more than our abilities.”',
|
||||
'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”',
|
||||
(...)]
|
||||
>>> response.xpath('//span[has-class("text")]/text()').getall()
|
||||
['“The world as we have created it is a process of our thinking. It cannot be changed without changing our thinking.”',
|
||||
'“It is our choices, Harry, that show what we truly are, far more than our abilities.”',
|
||||
'“There are only two ways to live your life. One is as though nothing is a miracle. The other is as though everything is a miracle.”',
|
||||
...]
|
||||
|
||||
And with one simple, cleverer XPath we are able to extract all quotes from
|
||||
the page. We could have constructed a loop over our first XPath to increase
|
||||
|
|
@ -132,7 +142,7 @@ a use case:
|
|||
|
||||
Say you want to find the ``Next`` button on the page. Type ``Next`` into the
|
||||
search bar on the top right of the `Inspector`. You should get two results.
|
||||
The first is a ``li`` tag with the ``class="text"``, the second the text
|
||||
The first is a ``li`` tag with the ``class="next"``, the second the text
|
||||
of an ``a`` tag. Right click on the ``a`` tag and select ``Scroll into View``.
|
||||
If you hover over the tag, you'll see the button highlighted. From here
|
||||
we could easily create a :ref:`Link Extractor <topics-link-extractors>` to
|
||||
|
|
@ -159,7 +169,11 @@ The page is quite similar to the basic `quotes.toscrape.com`_-page,
|
|||
but instead of the above-mentioned ``Next`` button, the page
|
||||
automatically loads new quotes when you scroll to the bottom. We
|
||||
could go ahead and try out different XPaths directly, but instead
|
||||
we'll check another quite useful command from the scrapy shell::
|
||||
we'll check another quite useful command from the Scrapy shell:
|
||||
|
||||
.. skip: next
|
||||
|
||||
.. code-block:: none
|
||||
|
||||
$ scrapy shell "quotes.toscrape.com/scroll"
|
||||
(...)
|
||||
|
|
|
|||
|
|
@ -199,7 +199,7 @@ CookiesMiddleware
|
|||
|
||||
This middleware enables working with sites that require cookies, such as
|
||||
those that use sessions. It keeps track of cookies sent by web servers, and
|
||||
send them back on subsequent requests (from that spider), just like web
|
||||
sends them back on subsequent requests (from that spider), just like web
|
||||
browsers do.
|
||||
|
||||
The following settings can be used to configure the cookie middleware:
|
||||
|
|
@ -259,8 +259,8 @@ COOKIES_DEBUG
|
|||
|
||||
Default: ``False``
|
||||
|
||||
If enabled, Scrapy will log all cookies sent in requests (ie. ``Cookie``
|
||||
header) and all cookies received in responses (ie. ``Set-Cookie`` header).
|
||||
If enabled, Scrapy will log all cookies sent in requests (i.e. ``Cookie``
|
||||
header) and all cookies received in responses (i.e. ``Set-Cookie`` header).
|
||||
|
||||
Here's an example of a log with :setting:`COOKIES_DEBUG` enabled::
|
||||
|
||||
|
|
@ -474,7 +474,7 @@ DBM storage backend
|
|||
|
||||
A DBM_ storage backend is also available for the HTTP cache middleware.
|
||||
|
||||
By default, it uses the anydbm_ module, but you can change it with the
|
||||
By default, it uses the :mod:`dbm`, but you can change it with the
|
||||
:setting:`HTTPCACHE_DBM_MODULE` setting.
|
||||
|
||||
.. _httpcache-storage-custom:
|
||||
|
|
@ -626,7 +626,7 @@ HTTPCACHE_DBM_MODULE
|
|||
|
||||
.. versionadded:: 0.13
|
||||
|
||||
Default: ``'anydbm'``
|
||||
Default: ``'dbm'``
|
||||
|
||||
The database module to use in the :ref:`DBM storage backend
|
||||
<httpcache-storage-dbm>`. This setting is specific to the DBM backend.
|
||||
|
|
@ -672,7 +672,7 @@ sometimes a more nuanced policy is desirable.
|
|||
|
||||
This setting still respects ``Cache-Control: no-store`` directives in responses.
|
||||
If you don't want that, filter ``no-store`` out of the Cache-Control headers in
|
||||
responses you feedto the cache middleware.
|
||||
responses you feed to the cache middleware.
|
||||
|
||||
.. setting:: HTTPCACHE_IGNORE_RESPONSE_CACHE_CONTROLS
|
||||
|
||||
|
|
@ -686,7 +686,7 @@ Default: ``[]``
|
|||
List of Cache-Control directives in responses to be ignored.
|
||||
|
||||
Sites often set "no-store", "no-cache", "must-revalidate", etc., but get
|
||||
upset at the traffic a spider can generate if it respects those
|
||||
upset at the traffic a spider can generate if it actually respects those
|
||||
directives. This allows to selectively ignore Cache-Control directives
|
||||
that are known to be unimportant for the sites being crawled.
|
||||
|
||||
|
|
@ -709,7 +709,7 @@ HttpCompressionMiddleware
|
|||
provided `brotlipy`_ is installed.
|
||||
|
||||
.. _brotli-compressed: https://www.ietf.org/rfc/rfc7932.txt
|
||||
.. _brotlipy: https://pypi.python.org/pypi/brotlipy
|
||||
.. _brotlipy: https://pypi.org/project/brotlipy/
|
||||
|
||||
HttpCompressionMiddleware Settings
|
||||
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||
|
|
@ -868,7 +868,7 @@ Whether the Meta Refresh middleware will be enabled.
|
|||
METAREFRESH_IGNORE_TAGS
|
||||
^^^^^^^^^^^^^^^^^^^^^^^
|
||||
|
||||
Default: ``['script', 'noscript']``
|
||||
Default: ``[]``
|
||||
|
||||
Meta tags within these tags are ignored.
|
||||
|
||||
|
|
@ -1038,7 +1038,7 @@ Based on `RobotFileParser
|
|||
* is Python's built-in robots.txt_ parser
|
||||
|
||||
* is compliant with `Martijn Koster's 1996 draft specification
|
||||
<http://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
<https://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
|
||||
* lacks support for wildcard matching
|
||||
|
||||
|
|
@ -1061,7 +1061,7 @@ Based on `Reppy <https://github.com/seomoz/reppy/>`_:
|
|||
<https://github.com/seomoz/rep-cpp>`_
|
||||
|
||||
* is compliant with `Martijn Koster's 1996 draft specification
|
||||
<http://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
<https://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
|
||||
* supports wildcard matching
|
||||
|
||||
|
|
@ -1086,7 +1086,7 @@ Based on `Robotexclusionrulesparser <http://nikitathespider.com/python/rerp/>`_:
|
|||
* implemented in Python
|
||||
|
||||
* is compliant with `Martijn Koster's 1996 draft specification
|
||||
<http://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
<https://www.robotstxt.org/norobots-rfc.txt>`_
|
||||
|
||||
* supports wildcard matching
|
||||
|
||||
|
|
@ -1115,7 +1115,7 @@ implementing the methods described below.
|
|||
.. autoclass:: RobotParser
|
||||
:members:
|
||||
|
||||
.. _robots.txt: http://www.robotstxt.org/
|
||||
.. _robots.txt: https://www.robotstxt.org/
|
||||
|
||||
DownloaderStats
|
||||
---------------
|
||||
|
|
@ -1155,7 +1155,7 @@ AjaxCrawlMiddleware
|
|||
|
||||
Middleware that finds 'AJAX crawlable' page variants based
|
||||
on meta-fragment html tag. See
|
||||
https://developers.google.com/webmasters/ajax-crawling/docs/getting-started
|
||||
https://developers.google.com/search/docs/ajax-crawling/docs/getting-started
|
||||
for more info.
|
||||
|
||||
.. note::
|
||||
|
|
@ -1202,4 +1202,3 @@ The default encoding for proxy authentication on :class:`HttpProxyMiddleware`.
|
|||
|
||||
|
||||
.. _DBM: https://en.wikipedia.org/wiki/Dbm
|
||||
.. _anydbm: https://docs.python.org/2/library/anydbm.html
|
||||
|
|
|
|||
|
|
@ -172,27 +172,27 @@ data from it:
|
|||
data in JSON format, which you can then parse with `json.loads`_.
|
||||
|
||||
For example, if the JavaScript code contains a separate line like
|
||||
``var data = {"field": "value"};`` you can extract that data as follows::
|
||||
``var data = {"field": "value"};`` you can extract that data as follows:
|
||||
|
||||
>>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n'
|
||||
>>> json_data = response.css('script::text').re_first(pattern)
|
||||
>>> json.loads(json_data)
|
||||
{'field': 'value'}
|
||||
>>> pattern = r'\bvar\s+data\s*=\s*(\{.*?\})\s*;\s*\n'
|
||||
>>> json_data = response.css('script::text').re_first(pattern)
|
||||
>>> json.loads(json_data)
|
||||
{'field': 'value'}
|
||||
|
||||
- Otherwise, use js2xml_ to convert the JavaScript code into an XML document
|
||||
that you can parse using :ref:`selectors <topics-selectors>`.
|
||||
|
||||
For example, if the JavaScript code contains
|
||||
``var data = {field: "value"};`` you can extract that data as follows::
|
||||
``var data = {field: "value"};`` you can extract that data as follows:
|
||||
|
||||
>>> import js2xml
|
||||
>>> import lxml.etree
|
||||
>>> from parsel import Selector
|
||||
>>> javascript = response.css('script::text').get()
|
||||
>>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode')
|
||||
>>> selector = Selector(text=xml)
|
||||
>>> selector.css('var[name="data"]').get()
|
||||
'<var name="data"><object><property name="field"><string>value</string></property></object></var>'
|
||||
>>> import js2xml
|
||||
>>> import lxml.etree
|
||||
>>> from parsel import Selector
|
||||
>>> javascript = response.css('script::text').get()
|
||||
>>> xml = lxml.etree.tostring(js2xml.parse(javascript), encoding='unicode')
|
||||
>>> selector = Selector(text=xml)
|
||||
>>> selector.css('var[name="data"]').get()
|
||||
'<var name="data"><object><property name="field"><string>value</string></property></object></var>'
|
||||
|
||||
.. _topics-javascript-rendering:
|
||||
|
||||
|
|
@ -241,12 +241,12 @@ along with `scrapy-selenium`_ for seamless integration.
|
|||
.. _headless browser: https://en.wikipedia.org/wiki/Headless_browser
|
||||
.. _JavaScript: https://en.wikipedia.org/wiki/JavaScript
|
||||
.. _js2xml: https://github.com/scrapinghub/js2xml
|
||||
.. _json.loads: https://docs.python.org/library/json.html#json.loads
|
||||
.. _json.loads: https://docs.python.org/3/library/json.html#json.loads
|
||||
.. _pytesseract: https://github.com/madmaze/pytesseract
|
||||
.. _regular expression: https://docs.python.org/library/re.html
|
||||
.. _regular expression: https://docs.python.org/3/library/re.html
|
||||
.. _scrapy-selenium: https://github.com/clemfromspace/scrapy-selenium
|
||||
.. _scrapy-splash: https://github.com/scrapy-plugins/scrapy-splash
|
||||
.. _Selenium: https://www.seleniumhq.org/
|
||||
.. _Selenium: https://www.selenium.dev/
|
||||
.. _Splash: https://github.com/scrapinghub/splash
|
||||
.. _tabula-py: https://github.com/chezou/tabula-py
|
||||
.. _wget: https://www.gnu.org/software/wget/
|
||||
|
|
|
|||
|
|
@ -9,19 +9,19 @@ Sending e-mail
|
|||
|
||||
Although Python makes sending e-mails relatively easy via the `smtplib`_
|
||||
library, Scrapy provides its own facility for sending e-mails which is very
|
||||
easy to use and it's implemented using `Twisted non-blocking IO`_, to avoid
|
||||
interfering with the non-blocking IO of the crawler. It also provides a
|
||||
simple API for sending attachments and it's very easy to configure, with a few
|
||||
:ref:`settings <topics-email-settings>`.
|
||||
easy to use and it's implemented using :doc:`Twisted non-blocking IO
|
||||
<twisted:core/howto/defer-intro>`, to avoid interfering with the non-blocking
|
||||
IO of the crawler. It also provides a simple API for sending attachments and
|
||||
it's very easy to configure, with a few :ref:`settings
|
||||
<topics-email-settings>`.
|
||||
|
||||
.. _smtplib: https://docs.python.org/2/library/smtplib.html
|
||||
.. _Twisted non-blocking IO: https://twistedmatrix.com/documents/current/core/howto/defer-intro.html
|
||||
|
||||
Quick example
|
||||
=============
|
||||
|
||||
There are two ways to instantiate the mail sender. You can instantiate it using
|
||||
the standard constructor::
|
||||
the standard ``__init__`` method::
|
||||
|
||||
from scrapy.mail import MailSender
|
||||
mailer = MailSender()
|
||||
|
|
@ -39,7 +39,8 @@ MailSender class reference
|
|||
==========================
|
||||
|
||||
MailSender is the preferred class to use for sending emails from Scrapy, as it
|
||||
uses `Twisted non-blocking IO`_, like the rest of the framework.
|
||||
uses :doc:`Twisted non-blocking IO <twisted:core/howto/defer-intro>`, like the
|
||||
rest of the framework.
|
||||
|
||||
.. class:: MailSender(smtphost=None, mailfrom=None, smtpuser=None, smtppass=None, smtpport=None)
|
||||
|
||||
|
|
@ -111,7 +112,7 @@ uses `Twisted non-blocking IO`_, like the rest of the framework.
|
|||
Mail settings
|
||||
=============
|
||||
|
||||
These settings define the default constructor values of the :class:`MailSender`
|
||||
These settings define the default ``__init__`` method values of the :class:`MailSender`
|
||||
class, and can be used to configure e-mail notifications in your project without
|
||||
writing any code (for those extensions and code that uses :class:`MailSender`).
|
||||
|
||||
|
|
|
|||
|
|
@ -87,8 +87,8 @@ described next.
|
|||
1. Declaring a serializer in the field
|
||||
--------------------------------------
|
||||
|
||||
If you use :class:`~.Item` you can declare a serializer in the
|
||||
:ref:`field metadata <topics-items-fields>`. The serializer must be
|
||||
If you use :class:`~.Item` you can declare a serializer in the
|
||||
:ref:`field metadata <topics-items-fields>`. The serializer must be
|
||||
a callable which receives a value and returns its serialized form.
|
||||
|
||||
Example::
|
||||
|
|
@ -144,7 +144,7 @@ BaseItemExporter
|
|||
defining what fields to export, whether to export empty fields, or which
|
||||
encoding to use.
|
||||
|
||||
These features can be configured through the constructor arguments which
|
||||
These features can be configured through the ``__init__`` method arguments which
|
||||
populate their respective instance attributes: :attr:`fields_to_export`,
|
||||
:attr:`export_empty_fields`, :attr:`encoding`, :attr:`indent`.
|
||||
|
||||
|
|
@ -246,8 +246,8 @@ XmlItemExporter
|
|||
:param item_element: The name of each item element in the exported XML.
|
||||
:type item_element: str
|
||||
|
||||
The additional keyword arguments of this constructor are passed to the
|
||||
:class:`BaseItemExporter` constructor.
|
||||
The additional keyword arguments of this ``__init__`` method are passed to the
|
||||
:class:`BaseItemExporter` ``__init__`` method.
|
||||
|
||||
A typical output of this exporter would be::
|
||||
|
||||
|
|
@ -306,9 +306,9 @@ CsvItemExporter
|
|||
multi-valued fields, if found.
|
||||
:type include_headers_line: str
|
||||
|
||||
The additional keyword arguments of this constructor are passed to the
|
||||
:class:`BaseItemExporter` constructor, and the leftover arguments to the
|
||||
`csv.writer`_ constructor, so you can use any ``csv.writer`` constructor
|
||||
The additional keyword arguments of this ``__init__`` method are passed to the
|
||||
:class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to the
|
||||
`csv.writer`_ ``__init__`` method, so you can use any ``csv.writer`` ``__init__`` method
|
||||
argument to customize this exporter.
|
||||
|
||||
A typical output of this exporter would be::
|
||||
|
|
@ -334,8 +334,8 @@ PickleItemExporter
|
|||
|
||||
For more information, refer to the `pickle module documentation`_.
|
||||
|
||||
The additional keyword arguments of this constructor are passed to the
|
||||
:class:`BaseItemExporter` constructor.
|
||||
The additional keyword arguments of this ``__init__`` method are passed to the
|
||||
:class:`BaseItemExporter` ``__init__`` method.
|
||||
|
||||
Pickle isn't a human readable format, so no output examples are provided.
|
||||
|
||||
|
|
@ -351,8 +351,8 @@ PprintItemExporter
|
|||
:param file: the file-like object to use for exporting the data. Its ``write`` method should
|
||||
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
|
||||
|
||||
The additional keyword arguments of this constructor are passed to the
|
||||
:class:`BaseItemExporter` constructor.
|
||||
The additional keyword arguments of this ``__init__`` method are passed to the
|
||||
:class:`BaseItemExporter` ``__init__`` method.
|
||||
|
||||
A typical output of this exporter would be::
|
||||
|
||||
|
|
@ -367,10 +367,10 @@ JsonItemExporter
|
|||
.. class:: JsonItemExporter(file, \**kwargs)
|
||||
|
||||
Exports Items in JSON format to the specified file-like object, writing all
|
||||
objects as a list of objects. The additional constructor arguments are
|
||||
passed to the :class:`BaseItemExporter` constructor, and the leftover
|
||||
arguments to the `JSONEncoder`_ constructor, so you can use any
|
||||
`JSONEncoder`_ constructor argument to customize this exporter.
|
||||
objects as a list of objects. The additional ``__init__`` method arguments are
|
||||
passed to the :class:`BaseItemExporter` ``__init__`` method, and the leftover
|
||||
arguments to the `JSONEncoder`_ ``__init__`` method, so you can use any
|
||||
`JSONEncoder`_ ``__init__`` method argument to customize this exporter.
|
||||
|
||||
:param file: the file-like object to use for exporting the data. Its ``write`` method should
|
||||
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
|
||||
|
|
@ -398,10 +398,10 @@ JsonLinesItemExporter
|
|||
.. class:: JsonLinesItemExporter(file, \**kwargs)
|
||||
|
||||
Exports Items in JSON format to the specified file-like object, writing one
|
||||
JSON-encoded item per line. The additional constructor arguments are passed
|
||||
to the :class:`BaseItemExporter` constructor, and the leftover arguments to
|
||||
the `JSONEncoder`_ constructor, so you can use any `JSONEncoder`_
|
||||
constructor argument to customize this exporter.
|
||||
JSON-encoded item per line. The additional ``__init__`` method arguments are passed
|
||||
to the :class:`BaseItemExporter` ``__init__`` method, and the leftover arguments to
|
||||
the `JSONEncoder`_ ``__init__`` method, so you can use any `JSONEncoder`_
|
||||
``__init__`` method argument to customize this exporter.
|
||||
|
||||
:param file: the file-like object to use for exporting the data. Its ``write`` method should
|
||||
accept ``bytes`` (a disk file opened in binary mode, a ``io.BytesIO`` object, etc)
|
||||
|
|
|
|||
|
|
@ -28,7 +28,7 @@ Loading & activating extensions
|
|||
|
||||
Extensions are loaded and activated at startup by instantiating a single
|
||||
instance of the extension class. Therefore, all the extension initialization
|
||||
code must be performed in the class constructor (``__init__`` method).
|
||||
code must be performed in the class ``__init__`` method.
|
||||
|
||||
To make an extension available, add it to the :setting:`EXTENSIONS` setting in
|
||||
your Scrapy settings. In :setting:`EXTENSIONS`, each extension is represented
|
||||
|
|
@ -63,7 +63,7 @@ but disabled unless the :setting:`HTTPCACHE_ENABLED` setting is set.
|
|||
Disabling an extension
|
||||
======================
|
||||
|
||||
In order to disable an extension that comes enabled by default (ie. those
|
||||
In order to disable an extension that comes enabled by default (i.e. those
|
||||
included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to
|
||||
``None``. For example::
|
||||
|
||||
|
|
@ -345,7 +345,7 @@ signal is received. The information dumped is the following:
|
|||
After the stack trace and engine status is dumped, the Scrapy process continues
|
||||
running normally.
|
||||
|
||||
This extension only works on POSIX-compliant platforms (ie. not Windows),
|
||||
This extension only works on POSIX-compliant platforms (i.e. not Windows),
|
||||
because the `SIGQUIT`_ and `SIGUSR2`_ signals are not available on Windows.
|
||||
|
||||
There are at least two ways to send Scrapy the `SIGQUIT`_ signal:
|
||||
|
|
@ -370,7 +370,7 @@ running normally.
|
|||
|
||||
For more info see `Debugging in Python`_.
|
||||
|
||||
This extension only works on POSIX-compliant platforms (ie. not Windows).
|
||||
This extension only works on POSIX-compliant platforms (i.e. not Windows).
|
||||
|
||||
.. _Python debugger: https://docs.python.org/2/library/pdb.html
|
||||
.. _Debugging in Python: https://pythonconquerstheuniverse.wordpress.com/2009/09/10/debugging-in-python/
|
||||
|
|
|
|||
|
|
@ -99,12 +99,12 @@ The storages backends supported out of the box are:
|
|||
|
||||
* :ref:`topics-feed-storage-fs`
|
||||
* :ref:`topics-feed-storage-ftp`
|
||||
* :ref:`topics-feed-storage-s3` (requires botocore_ or boto_)
|
||||
* :ref:`topics-feed-storage-s3` (requires botocore_)
|
||||
* :ref:`topics-feed-storage-stdout`
|
||||
|
||||
Some storage backends may be unavailable if the required external libraries are
|
||||
not available. For example, the S3 backend is only available if the botocore_
|
||||
or boto_ library is installed (Scrapy supports boto_ only on Python 2).
|
||||
library is installed.
|
||||
|
||||
|
||||
.. _topics-feed-uri-params:
|
||||
|
|
@ -182,7 +182,7 @@ The feeds are stored on `Amazon S3`_.
|
|||
* ``s3://mybucket/path/to/export.csv``
|
||||
* ``s3://aws_key:aws_secret@mybucket/path/to/export.csv``
|
||||
|
||||
* Required external libraries: `botocore`_ (Python 2 and Python 3) or `boto`_ (Python 2 only)
|
||||
* Required external libraries: `botocore`_
|
||||
|
||||
The AWS credentials can be passed as user/password in the URI, or they can be
|
||||
passed through the following settings:
|
||||
|
|
@ -301,7 +301,7 @@ FEED_STORE_EMPTY
|
|||
|
||||
Default: ``False``
|
||||
|
||||
Whether to export empty feeds (ie. feeds with no items).
|
||||
Whether to export empty feeds (i.e. feeds with no items).
|
||||
|
||||
.. setting:: FEED_STORAGES
|
||||
|
||||
|
|
@ -399,6 +399,5 @@ format in :setting:`FEED_EXPORTERS`. E.g., to disable the built-in CSV exporter
|
|||
|
||||
.. _URI: https://en.wikipedia.org/wiki/Uniform_Resource_Identifier
|
||||
.. _Amazon S3: https://aws.amazon.com/s3/
|
||||
.. _boto: https://github.com/boto/boto
|
||||
.. _botocore: https://github.com/boto/botocore
|
||||
.. _Canned ACL: https://docs.aws.amazon.com/AmazonS3/latest/dev/acl-overview.html#canned-acl
|
||||
|
|
|
|||
|
|
@ -29,7 +29,8 @@ Each item pipeline component is a Python class that must implement the following
|
|||
|
||||
This method is called for every item pipeline component. :meth:`process_item`
|
||||
must either: return a dict with data, return an :class:`~scrapy.item.Item`
|
||||
(or any descendant class) object, return a `Twisted Deferred`_ or raise
|
||||
(or any descendant class) object, return a
|
||||
:class:`~twisted.internet.defer.Deferred` or raise
|
||||
:exc:`~scrapy.exceptions.DropItem` exception. Dropped items are no longer
|
||||
processed by further pipeline components.
|
||||
|
||||
|
|
@ -67,8 +68,6 @@ Additionally, they may also implement the following methods:
|
|||
:type crawler: :class:`~scrapy.crawler.Crawler` object
|
||||
|
||||
|
||||
.. _Twisted Deferred: https://twistedmatrix.com/documents/current/core/howto/defer.html
|
||||
|
||||
Item pipeline example
|
||||
=====================
|
||||
|
||||
|
|
@ -159,14 +158,15 @@ method and how to clean up the resources properly.::
|
|||
self.db[self.collection_name].insert_one(dict(item))
|
||||
return item
|
||||
|
||||
.. _MongoDB: https://www.mongodb.org/
|
||||
.. _pymongo: https://api.mongodb.org/python/current/
|
||||
.. _MongoDB: https://www.mongodb.com/
|
||||
.. _pymongo: https://api.mongodb.com/python/current/
|
||||
|
||||
|
||||
Take screenshot of item
|
||||
-----------------------
|
||||
|
||||
This example demonstrates how to return Deferred_ from :meth:`process_item` method.
|
||||
This example demonstrates how to return a
|
||||
:class:`~twisted.internet.defer.Deferred` from the :meth:`process_item` method.
|
||||
It uses Splash_ to render screenshot of item url. Pipeline
|
||||
makes request to locally running instance of Splash_. After request is downloaded
|
||||
and Deferred callback fires, it saves item to a file and adds filename to an item.
|
||||
|
|
@ -209,7 +209,6 @@ and Deferred callback fires, it saves item to a file and adds filename to an ite
|
|||
return item
|
||||
|
||||
.. _Splash: https://splash.readthedocs.io/en/stable/
|
||||
.. _Deferred: https://twistedmatrix.com/documents/current/core/howto/defer.html
|
||||
|
||||
Duplicates filter
|
||||
-----------------
|
||||
|
|
|
|||
|
|
@ -16,12 +16,12 @@ especially in a larger project with many spiders.
|
|||
To define common output data format Scrapy provides the :class:`Item` class.
|
||||
:class:`Item` objects are simple containers used to collect the scraped data.
|
||||
They provide a `dictionary-like`_ API with a convenient syntax for declaring
|
||||
their available fields.
|
||||
their available fields.
|
||||
|
||||
Various Scrapy components use extra information provided by Items:
|
||||
Various Scrapy components use extra information provided by Items:
|
||||
exporters look at declared fields to figure out columns to export,
|
||||
serialization can be customized using Item fields metadata, :mod:`trackref`
|
||||
tracks Item instances to help find memory leaks
|
||||
tracks Item instances to help find memory leaks
|
||||
(see :ref:`topics-leaks-trackrefs`), etc.
|
||||
|
||||
.. _dictionary-like: https://docs.python.org/2/library/stdtypes.html#dict
|
||||
|
|
@ -84,77 +84,74 @@ notice the API is very similar to the `dict API`_.
|
|||
Creating items
|
||||
--------------
|
||||
|
||||
::
|
||||
>>> product = Product(name='Desktop PC', price=1000)
|
||||
>>> print(product)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
>>> product = Product(name='Desktop PC', price=1000)
|
||||
>>> print(product)
|
||||
Product(name='Desktop PC', price=1000)
|
||||
|
||||
Getting field values
|
||||
--------------------
|
||||
|
||||
::
|
||||
>>> product['name']
|
||||
Desktop PC
|
||||
>>> product.get('name')
|
||||
Desktop PC
|
||||
|
||||
>>> product['name']
|
||||
Desktop PC
|
||||
>>> product.get('name')
|
||||
Desktop PC
|
||||
>>> product['price']
|
||||
1000
|
||||
|
||||
>>> product['price']
|
||||
1000
|
||||
>>> product['last_updated']
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'last_updated'
|
||||
|
||||
>>> product['last_updated']
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'last_updated'
|
||||
>>> product.get('last_updated', 'not set')
|
||||
not set
|
||||
|
||||
>>> product.get('last_updated', 'not set')
|
||||
not set
|
||||
>>> product['lala'] # getting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'lala'
|
||||
|
||||
>>> product['lala'] # getting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'lala'
|
||||
>>> product.get('lala', 'unknown field')
|
||||
'unknown field'
|
||||
|
||||
>>> product.get('lala', 'unknown field')
|
||||
'unknown field'
|
||||
>>> 'name' in product # is name field populated?
|
||||
True
|
||||
|
||||
>>> 'name' in product # is name field populated?
|
||||
True
|
||||
>>> 'last_updated' in product # is last_updated populated?
|
||||
False
|
||||
|
||||
>>> 'last_updated' in product # is last_updated populated?
|
||||
False
|
||||
>>> 'last_updated' in product.fields # is last_updated a declared field?
|
||||
True
|
||||
|
||||
>>> 'last_updated' in product.fields # is last_updated a declared field?
|
||||
True
|
||||
>>> 'lala' in product.fields # is lala a declared field?
|
||||
False
|
||||
|
||||
>>> 'lala' in product.fields # is lala a declared field?
|
||||
False
|
||||
|
||||
Setting field values
|
||||
--------------------
|
||||
|
||||
::
|
||||
>>> product['last_updated'] = 'today'
|
||||
>>> product['last_updated']
|
||||
today
|
||||
|
||||
>>> product['last_updated'] = 'today'
|
||||
>>> product['last_updated']
|
||||
today
|
||||
>>> product['lala'] = 'test' # setting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
>>> product['lala'] = 'test' # setting unknown field
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
Accessing all populated values
|
||||
------------------------------
|
||||
|
||||
To access all populated values, just use the typical `dict API`_::
|
||||
To access all populated values, just use the typical `dict API`_:
|
||||
|
||||
>>> product.keys()
|
||||
['price', 'name']
|
||||
>>> product.keys()
|
||||
['price', 'name']
|
||||
|
||||
>>> product.items()
|
||||
[('price', 1000), ('name', 'Desktop PC')]
|
||||
>>> product.items()
|
||||
[('price', 1000), ('name', 'Desktop PC')]
|
||||
|
||||
|
||||
.. _copying-items:
|
||||
|
|
@ -169,7 +166,7 @@ If your item contains mutable_ values like lists or dictionaries, a shallow
|
|||
copy will keep references to the same mutable values across all different
|
||||
copies.
|
||||
|
||||
.. _mutable: https://docs.python.org/glossary.html#term-mutable
|
||||
.. _mutable: https://docs.python.org/3/glossary.html#term-mutable
|
||||
|
||||
For example, if you have an item with a list of tags, and you create a shallow
|
||||
copy of that item, both the original item and the copy have the same list of
|
||||
|
|
@ -180,7 +177,7 @@ If that is not the desired behavior, use a deep copy instead.
|
|||
|
||||
See the `documentation of the copy module`_ for more information.
|
||||
|
||||
.. _documentation of the copy module: https://docs.python.org/library/copy.html
|
||||
.. _documentation of the copy module: https://docs.python.org/3/library/copy.html
|
||||
|
||||
To create a shallow copy of an item, you can either call
|
||||
:meth:`~scrapy.item.Item.copy` on an existing item
|
||||
|
|
@ -194,20 +191,21 @@ To create a deep copy, call :meth:`~scrapy.item.Item.deepcopy` instead
|
|||
Other common tasks
|
||||
------------------
|
||||
|
||||
Creating dicts from items::
|
||||
Creating dicts from items:
|
||||
|
||||
>>> dict(product) # create a dict from all populated values
|
||||
{'price': 1000, 'name': 'Desktop PC'}
|
||||
>>> dict(product) # create a dict from all populated values
|
||||
{'price': 1000, 'name': 'Desktop PC'}
|
||||
|
||||
Creating items from dicts::
|
||||
Creating items from dicts:
|
||||
|
||||
>>> Product({'name': 'Laptop PC', 'price': 1500})
|
||||
Product(price=1500, name='Laptop PC')
|
||||
>>> Product({'name': 'Laptop PC', 'price': 1500})
|
||||
Product(price=1500, name='Laptop PC')
|
||||
|
||||
>>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
>>> Product({'name': 'Laptop PC', 'lala': 1500}) # warning: unknown field in dict
|
||||
Traceback (most recent call last):
|
||||
...
|
||||
KeyError: 'Product does not support field: lala'
|
||||
|
||||
Extending Items
|
||||
===============
|
||||
|
|
@ -237,7 +235,7 @@ Item objects
|
|||
|
||||
Return a new Item optionally initialized from the given argument.
|
||||
|
||||
Items replicate the standard `dict API`_, including its constructor, and
|
||||
Items replicate the standard `dict API`_, including its ``__init__`` method, and
|
||||
also provide the following additional API members:
|
||||
|
||||
.. automethod:: copy
|
||||
|
|
|
|||
|
|
@ -22,7 +22,7 @@ Job directory
|
|||
|
||||
To enable persistence support you just need to define a *job directory* through
|
||||
the ``JOBDIR`` setting. This directory will be for storing all required data to
|
||||
keep the state of a single job (ie. a spider run). It's important to note that
|
||||
keep the state of a single job (i.e. a spider run). It's important to note that
|
||||
this directory must not be shared by different spiders, or even different
|
||||
jobs/runs of the same spider, as it's meant to be used for storing the state of
|
||||
a *single* job.
|
||||
|
|
@ -71,34 +71,11 @@ on cookies.
|
|||
Request serialization
|
||||
---------------------
|
||||
|
||||
Requests must be serializable by the ``pickle`` module, in order for persistence
|
||||
to work, so you should make sure that your requests are serializable.
|
||||
|
||||
The most common issue here is to use ``lambda`` functions on request callbacks that
|
||||
can't be persisted.
|
||||
|
||||
So, for example, this won't work::
|
||||
|
||||
def some_callback(self, response):
|
||||
somearg = 'test'
|
||||
return scrapy.Request('http://www.example.com',
|
||||
callback=lambda r: self.other_callback(r, somearg))
|
||||
|
||||
def other_callback(self, response, somearg):
|
||||
print("the argument passed is: %s" % somearg)
|
||||
|
||||
But this will::
|
||||
|
||||
def some_callback(self, response):
|
||||
somearg = 'test'
|
||||
return scrapy.Request('http://www.example.com',
|
||||
callback=self.other_callback, cb_kwargs={'somearg': somearg})
|
||||
|
||||
def other_callback(self, response, somearg):
|
||||
print("the argument passed is: %s" % somearg)
|
||||
For persistence to work, :class:`~scrapy.http.Request` objects must be
|
||||
serializable with :mod:`pickle`, except for the ``callback`` and ``errback``
|
||||
values passed to their ``__init__`` method, which must be methods of the
|
||||
running :class:`~scrapy.spiders.Spider` class.
|
||||
|
||||
If you wish to log the requests that couldn't be serialized, you can set the
|
||||
:setting:`SCHEDULER_DEBUG` setting to ``True`` in the project's settings page.
|
||||
It is ``False`` by default.
|
||||
|
||||
.. _pickle: https://docs.python.org/library/pickle.html
|
||||
|
|
|
|||
|
|
@ -110,7 +110,7 @@ ties the response lifetime to the requests' one, and that would definitely
|
|||
cause memory leaks.
|
||||
|
||||
Let's see how we can discover the cause (without knowing it
|
||||
a-priori, of course) by using the ``trackref`` tool.
|
||||
a priori, of course) by using the ``trackref`` tool.
|
||||
|
||||
After the crawler is running for a few minutes and we notice its memory usage
|
||||
has grown a lot, we can enter its telnet console and check the live
|
||||
|
|
@ -132,21 +132,21 @@ and check the code of the spider to discover the nasty line that is
|
|||
generating the leaks (passing response references inside requests).
|
||||
|
||||
Sometimes extra information about live objects can be helpful.
|
||||
Let's check the oldest response::
|
||||
Let's check the oldest response:
|
||||
|
||||
>>> from scrapy.utils.trackref import get_oldest
|
||||
>>> r = get_oldest('HtmlResponse')
|
||||
>>> r.url
|
||||
'http://www.somenastyspider.com/product.php?pid=123'
|
||||
>>> from scrapy.utils.trackref import get_oldest
|
||||
>>> r = get_oldest('HtmlResponse')
|
||||
>>> r.url
|
||||
'http://www.somenastyspider.com/product.php?pid=123'
|
||||
|
||||
If you want to iterate over all objects, instead of getting the oldest one, you
|
||||
can use the :func:`scrapy.utils.trackref.iter_all` function::
|
||||
can use the :func:`scrapy.utils.trackref.iter_all` function:
|
||||
|
||||
>>> from scrapy.utils.trackref import iter_all
|
||||
>>> [r.url for r in iter_all('HtmlResponse')]
|
||||
['http://www.somenastyspider.com/product.php?pid=123',
|
||||
'http://www.somenastyspider.com/product.php?pid=584',
|
||||
...
|
||||
>>> from scrapy.utils.trackref import iter_all
|
||||
>>> [r.url for r in iter_all('HtmlResponse')]
|
||||
['http://www.somenastyspider.com/product.php?pid=123',
|
||||
'http://www.somenastyspider.com/product.php?pid=584',
|
||||
...]
|
||||
|
||||
Too many spiders?
|
||||
-----------------
|
||||
|
|
@ -155,10 +155,10 @@ If your project has too many spiders executed in parallel,
|
|||
the output of :func:`prefs()` can be difficult to read.
|
||||
For this reason, that function has a ``ignore`` argument which can be used to
|
||||
ignore a particular class (and all its subclases). For
|
||||
example, this won't show any live references to spiders::
|
||||
example, this won't show any live references to spiders:
|
||||
|
||||
>>> from scrapy.spiders import Spider
|
||||
>>> prefs(ignore=Spider)
|
||||
>>> from scrapy.spiders import Spider
|
||||
>>> prefs(ignore=Spider)
|
||||
|
||||
.. module:: scrapy.utils.trackref
|
||||
:synopsis: Track references of live objects
|
||||
|
|
@ -206,7 +206,7 @@ objects. If this is your case, and you can't find your leaks using ``trackref``,
|
|||
you still have another resource: the `Guppy library`_.
|
||||
If you're using Python3, see :ref:`topics-leaks-muppy`.
|
||||
|
||||
.. _Guppy library: https://pypi.python.org/pypi/guppy
|
||||
.. _Guppy library: https://pypi.org/project/guppy/
|
||||
|
||||
If you use ``pip``, you can install Guppy with the following command::
|
||||
|
||||
|
|
@ -214,41 +214,41 @@ If you use ``pip``, you can install Guppy with the following command::
|
|||
|
||||
The telnet console also comes with a built-in shortcut (``hpy``) for accessing
|
||||
Guppy heap objects. Here's an example to view all Python objects available in
|
||||
the heap using Guppy::
|
||||
the heap using Guppy:
|
||||
|
||||
>>> x = hpy.heap()
|
||||
>>> x.bytype
|
||||
Partition of a set of 297033 objects. Total size = 52587824 bytes.
|
||||
Index Count % Size % Cumulative % Type
|
||||
0 22307 8 16423880 31 16423880 31 dict
|
||||
1 122285 41 12441544 24 28865424 55 str
|
||||
2 68346 23 5966696 11 34832120 66 tuple
|
||||
3 227 0 5836528 11 40668648 77 unicode
|
||||
4 2461 1 2222272 4 42890920 82 type
|
||||
5 16870 6 2024400 4 44915320 85 function
|
||||
6 13949 5 1673880 3 46589200 89 types.CodeType
|
||||
7 13422 5 1653104 3 48242304 92 list
|
||||
8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern
|
||||
9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers
|
||||
<1676 more rows. Type e.g. '_.more' to view.>
|
||||
>>> x = hpy.heap()
|
||||
>>> x.bytype
|
||||
Partition of a set of 297033 objects. Total size = 52587824 bytes.
|
||||
Index Count % Size % Cumulative % Type
|
||||
0 22307 8 16423880 31 16423880 31 dict
|
||||
1 122285 41 12441544 24 28865424 55 str
|
||||
2 68346 23 5966696 11 34832120 66 tuple
|
||||
3 227 0 5836528 11 40668648 77 unicode
|
||||
4 2461 1 2222272 4 42890920 82 type
|
||||
5 16870 6 2024400 4 44915320 85 function
|
||||
6 13949 5 1673880 3 46589200 89 types.CodeType
|
||||
7 13422 5 1653104 3 48242304 92 list
|
||||
8 3735 1 1173680 2 49415984 94 _sre.SRE_Pattern
|
||||
9 1209 0 456936 1 49872920 95 scrapy.http.headers.Headers
|
||||
<1676 more rows. Type e.g. '_.more' to view.>
|
||||
|
||||
You can see that most space is used by dicts. Then, if you want to see from
|
||||
which attribute those dicts are referenced, you could do::
|
||||
which attribute those dicts are referenced, you could do:
|
||||
|
||||
>>> x.bytype[0].byvia
|
||||
Partition of a set of 22307 objects. Total size = 16423880 bytes.
|
||||
Index Count % Size % Cumulative % Referred Via:
|
||||
0 10982 49 9416336 57 9416336 57 '.__dict__'
|
||||
1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals'
|
||||
2 3097 14 1122904 7 13220744 80
|
||||
3 990 4 277200 2 13497944 82 "['cookies']"
|
||||
4 987 4 276360 2 13774304 84 "['cache']"
|
||||
5 985 4 275800 2 14050104 86 "['meta']"
|
||||
6 897 4 251160 2 14301264 87 '[2]'
|
||||
7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']"
|
||||
8 672 3 188160 1 14686312 89 "['cb_kwargs']"
|
||||
9 27 0 155016 1 14841328 90 '[1]'
|
||||
<333 more rows. Type e.g. '_.more' to view.>
|
||||
>>> x.bytype[0].byvia
|
||||
Partition of a set of 22307 objects. Total size = 16423880 bytes.
|
||||
Index Count % Size % Cumulative % Referred Via:
|
||||
0 10982 49 9416336 57 9416336 57 '.__dict__'
|
||||
1 1820 8 2681504 16 12097840 74 '.__dict__', '.func_globals'
|
||||
2 3097 14 1122904 7 13220744 80
|
||||
3 990 4 277200 2 13497944 82 "['cookies']"
|
||||
4 987 4 276360 2 13774304 84 "['cache']"
|
||||
5 985 4 275800 2 14050104 86 "['meta']"
|
||||
6 897 4 251160 2 14301264 87 '[2]'
|
||||
7 1 0 196888 1 14498152 88 "['moduleDict']", "['modules']"
|
||||
8 672 3 188160 1 14686312 89 "['cb_kwargs']"
|
||||
9 27 0 155016 1 14841328 90 '[1]'
|
||||
<333 more rows. Type e.g. '_.more' to view.>
|
||||
|
||||
As you can see, the Guppy module is very powerful but also requires some deep
|
||||
knowledge about Python internals. For more info about Guppy, refer to the
|
||||
|
|
@ -260,7 +260,7 @@ knowledge about Python internals. For more info about Guppy, refer to the
|
|||
|
||||
Debugging memory leaks with muppy
|
||||
=================================
|
||||
If you're using Python 3, you can use muppy from `Pympler`_.
|
||||
You can use muppy from `Pympler`_.
|
||||
|
||||
.. _Pympler: https://pypi.org/project/Pympler/
|
||||
|
||||
|
|
@ -269,32 +269,32 @@ If you use ``pip``, you can install muppy with the following command::
|
|||
pip install Pympler
|
||||
|
||||
Here's an example to view all Python objects available in
|
||||
the heap using muppy::
|
||||
the heap using muppy:
|
||||
|
||||
>>> from pympler import muppy
|
||||
>>> all_objects = muppy.get_objects()
|
||||
>>> len(all_objects)
|
||||
28667
|
||||
>>> from pympler import summary
|
||||
>>> suml = summary.summarize(all_objects)
|
||||
>>> summary.print_(suml)
|
||||
types | # objects | total size
|
||||
==================================== | =========== | ============
|
||||
<class 'str | 9822 | 1.10 MB
|
||||
<class 'dict | 1658 | 856.62 KB
|
||||
<class 'type | 436 | 443.60 KB
|
||||
<class 'code | 2974 | 419.56 KB
|
||||
<class '_io.BufferedWriter | 2 | 256.34 KB
|
||||
<class 'set | 420 | 159.88 KB
|
||||
<class '_io.BufferedReader | 1 | 128.17 KB
|
||||
<class 'wrapper_descriptor | 1130 | 88.28 KB
|
||||
<class 'tuple | 1304 | 86.57 KB
|
||||
<class 'weakref | 1013 | 79.14 KB
|
||||
<class 'builtin_function_or_method | 958 | 67.36 KB
|
||||
<class 'method_descriptor | 865 | 60.82 KB
|
||||
<class 'abc.ABCMeta | 62 | 59.96 KB
|
||||
<class 'list | 446 | 58.52 KB
|
||||
<class 'int | 1425 | 43.20 KB
|
||||
>>> from pympler import muppy
|
||||
>>> all_objects = muppy.get_objects()
|
||||
>>> len(all_objects)
|
||||
28667
|
||||
>>> from pympler import summary
|
||||
>>> suml = summary.summarize(all_objects)
|
||||
>>> summary.print_(suml)
|
||||
types | # objects | total size
|
||||
==================================== | =========== | ============
|
||||
<class 'str | 9822 | 1.10 MB
|
||||
<class 'dict | 1658 | 856.62 KB
|
||||
<class 'type | 436 | 443.60 KB
|
||||
<class 'code | 2974 | 419.56 KB
|
||||
<class '_io.BufferedWriter | 2 | 256.34 KB
|
||||
<class 'set | 420 | 159.88 KB
|
||||
<class '_io.BufferedReader | 1 | 128.17 KB
|
||||
<class 'wrapper_descriptor | 1130 | 88.28 KB
|
||||
<class 'tuple | 1304 | 86.57 KB
|
||||
<class 'weakref | 1013 | 79.14 KB
|
||||
<class 'builtin_function_or_method | 958 | 67.36 KB
|
||||
<class 'method_descriptor | 865 | 60.82 KB
|
||||
<class 'abc.ABCMeta | 62 | 59.96 KB
|
||||
<class 'list | 446 | 58.52 KB
|
||||
<class 'int | 1425 | 43.20 KB
|
||||
|
||||
For more info about muppy, refer to the `muppy documentation`_.
|
||||
|
||||
|
|
@ -311,9 +311,9 @@ though neither Scrapy nor your project are leaking memory. This is due to a
|
|||
(not so well) known problem of Python, which may not return released memory to
|
||||
the operating system in some cases. For more information on this issue see:
|
||||
|
||||
* `Python Memory Management <http://www.evanjones.ca/python-memory.html>`_
|
||||
* `Python Memory Management Part 2 <http://www.evanjones.ca/python-memory-part2.html>`_
|
||||
* `Python Memory Management Part 3 <http://www.evanjones.ca/python-memory-part3.html>`_
|
||||
* `Python Memory Management <https://www.evanjones.ca/python-memory.html>`_
|
||||
* `Python Memory Management Part 2 <https://www.evanjones.ca/python-memory-part2.html>`_
|
||||
* `Python Memory Management Part 3 <https://www.evanjones.ca/python-memory-part3.html>`_
|
||||
|
||||
The improvements proposed by Evan Jones, which are detailed in `this paper`_,
|
||||
got merged in Python 2.5, but this only reduces the problem, it doesn't fix it
|
||||
|
|
@ -327,7 +327,7 @@ completely. To quote the paper:
|
|||
to move to a compacting garbage collector, which is able to move objects in
|
||||
memory. This would require significant changes to the Python interpreter.*
|
||||
|
||||
.. _this paper: http://www.evanjones.ca/memoryallocator/
|
||||
.. _this paper: https://www.evanjones.ca/memoryallocator/
|
||||
|
||||
To keep memory consumption reasonable you can split the job into several
|
||||
smaller jobs or enable :ref:`persistent job queue <topics-jobs>`
|
||||
|
|
|
|||
|
|
@ -4,46 +4,33 @@
|
|||
Link Extractors
|
||||
===============
|
||||
|
||||
Link extractors are objects whose only purpose is to extract links from web
|
||||
pages (:class:`scrapy.http.Response` objects) which will be eventually
|
||||
followed.
|
||||
A link extractor is an object that extracts links from responses.
|
||||
|
||||
There is ``scrapy.linkextractors.LinkExtractor`` available
|
||||
in Scrapy, but you can create your own custom Link Extractors to suit your
|
||||
needs by implementing a simple interface.
|
||||
|
||||
The only public method that every link extractor has is ``extract_links``,
|
||||
which receives a :class:`~scrapy.http.Response` object and returns a list
|
||||
of :class:`scrapy.link.Link` objects. Link extractors are meant to be
|
||||
instantiated once and their ``extract_links`` method called several times
|
||||
with different responses to extract links to follow.
|
||||
|
||||
Link extractors are used in the :class:`~scrapy.spiders.CrawlSpider`
|
||||
class (available in Scrapy), through a set of rules, but you can also use it in
|
||||
your spiders, even if you don't subclass from
|
||||
:class:`~scrapy.spiders.CrawlSpider`, as its purpose is very simple: to
|
||||
extract links.
|
||||
The ``__init__`` method of
|
||||
:class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor` takes settings that
|
||||
determine which links may be extracted. :class:`LxmlLinkExtractor.extract_links
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` returns a
|
||||
list of matching :class:`scrapy.link.Link` objects from a
|
||||
:class:`~scrapy.http.Response` object.
|
||||
|
||||
Link extractors are used in :class:`~scrapy.spiders.CrawlSpider` spiders
|
||||
through a set of :class:`~scrapy.spiders.Rule` objects. You can also use link
|
||||
extractors in regular spiders.
|
||||
|
||||
.. _topics-link-extractors-ref:
|
||||
|
||||
Built-in link extractors reference
|
||||
==================================
|
||||
Link extractor reference
|
||||
========================
|
||||
|
||||
.. module:: scrapy.linkextractors
|
||||
:synopsis: Link extractors classes
|
||||
|
||||
Link extractors classes bundled with Scrapy are provided in the
|
||||
:mod:`scrapy.linkextractors` module.
|
||||
|
||||
The default link extractor is ``LinkExtractor``, which is the same as
|
||||
:class:`~.LxmlLinkExtractor`::
|
||||
The link extractor class is
|
||||
:class:`scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`. For convenience it
|
||||
can also be imported as ``scrapy.linkextractors.LinkExtractor``::
|
||||
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
|
||||
There used to be other link extractor classes in previous Scrapy versions,
|
||||
but they are deprecated now.
|
||||
|
||||
LxmlLinkExtractor
|
||||
-----------------
|
||||
|
||||
|
|
@ -62,7 +49,7 @@ LxmlLinkExtractor
|
|||
:type allow: a regular expression (or list of)
|
||||
|
||||
:param deny: a single regular expression (or list of regular expressions)
|
||||
that the (absolute) urls must match in order to be excluded (ie. not
|
||||
that the (absolute) urls must match in order to be excluded (i.e. not
|
||||
extracted). It has precedence over the ``allow`` parameter. If not
|
||||
given (or empty) it won't exclude any links.
|
||||
:type deny: a regular expression (or list of)
|
||||
|
|
@ -152,4 +139,6 @@ LxmlLinkExtractor
|
|||
from elements or attributes which allow leading/trailing whitespaces).
|
||||
:type strip: boolean
|
||||
|
||||
.. automethod:: extract_links
|
||||
|
||||
.. _scrapy.linkextractors: https://github.com/scrapy/scrapy/blob/master/scrapy/linkextractors/__init__.py
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ Using Item Loaders to populate items
|
|||
|
||||
To use an Item Loader, you must first instantiate it. You can either
|
||||
instantiate it with a dict-like object (e.g. Item or dict) or without one, in
|
||||
which case an Item is automatically instantiated in the Item Loader constructor
|
||||
which case an Item is automatically instantiated in the Item Loader ``__init__`` method
|
||||
using the Item class specified in the :attr:`ItemLoader.default_item_class`
|
||||
attribute.
|
||||
|
||||
|
|
@ -142,20 +142,6 @@ accept one (and only one) positional argument, which will be an iterable.
|
|||
containing the collected values (for that field). The result of the output
|
||||
processors is the value that will be finally assigned to the item.
|
||||
|
||||
If you want to use a plain function as a processor, make sure it receives
|
||||
``self`` as the first argument::
|
||||
|
||||
def lowercase_processor(self, values):
|
||||
for v in values:
|
||||
yield v.lower()
|
||||
|
||||
class MyItemLoader(ItemLoader):
|
||||
name_in = lowercase_processor
|
||||
|
||||
This is because whenever a function is assigned as a class variable, it becomes
|
||||
a method and would be passed the instance as the the first argument when being
|
||||
called. See `this answer on stackoverflow`_ for more details.
|
||||
|
||||
The other thing you need to keep in mind is that the values returned by input
|
||||
processors are collected internally (in lists) and then passed to output
|
||||
processors to populate the fields.
|
||||
|
|
@ -163,7 +149,7 @@ processors to populate the fields.
|
|||
Last, but not least, Scrapy comes with some :ref:`commonly used processors
|
||||
<topics-loaders-available-processors>` built-in for convenience.
|
||||
|
||||
.. _this answer on stackoverflow: https://stackoverflow.com/a/35322635
|
||||
|
||||
|
||||
Declaring Item Loaders
|
||||
======================
|
||||
|
|
@ -220,14 +206,12 @@ metadata. Here is an example::
|
|||
output_processor=TakeFirst(),
|
||||
)
|
||||
|
||||
::
|
||||
|
||||
>>> from scrapy.loader import ItemLoader
|
||||
>>> il = ItemLoader(item=Product())
|
||||
>>> il.add_value('name', [u'Welcome to my', u'<strong>website</strong>'])
|
||||
>>> il.add_value('price', [u'€', u'<span>1000</span>'])
|
||||
>>> il.load_item()
|
||||
{'name': u'Welcome to my website', 'price': u'1000'}
|
||||
>>> from scrapy.loader import ItemLoader
|
||||
>>> il = ItemLoader(item=Product())
|
||||
>>> il.add_value('name', [u'Welcome to my', u'<strong>website</strong>'])
|
||||
>>> il.add_value('price', [u'€', u'<span>1000</span>'])
|
||||
>>> il.load_item()
|
||||
{'name': u'Welcome to my website', 'price': u'1000'}
|
||||
|
||||
The precedence order, for both input and output processors, is as follows:
|
||||
|
||||
|
|
@ -271,7 +255,7 @@ There are several ways to modify Item Loader context values:
|
|||
loader.context['unit'] = 'cm'
|
||||
|
||||
2. On Item Loader instantiation (the keyword arguments of Item Loader
|
||||
constructor are stored in the Item Loader context)::
|
||||
``__init__`` method are stored in the Item Loader context)::
|
||||
|
||||
loader = ItemLoader(product, unit='cm')
|
||||
|
||||
|
|
@ -328,11 +312,11 @@ ItemLoader objects
|
|||
applied before processors
|
||||
:type re: str or compiled regex
|
||||
|
||||
Examples::
|
||||
Examples:
|
||||
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)')
|
||||
'FOO`
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> loader.get_value(u'name: foo', TakeFirst(), unicode.upper, re='name: (.+)')
|
||||
'FOO`
|
||||
|
||||
.. method:: add_value(field_name, value, \*processors, \**kwargs)
|
||||
|
||||
|
|
@ -491,6 +475,8 @@ ItemLoader objects
|
|||
.. attribute:: item
|
||||
|
||||
The :class:`~scrapy.item.Item` object being parsed by this Item Loader.
|
||||
This is mostly used as a property so when attempting to override this
|
||||
value, you may want to check out :attr:`default_item_class` first.
|
||||
|
||||
.. attribute:: context
|
||||
|
||||
|
|
@ -500,7 +486,7 @@ ItemLoader objects
|
|||
.. attribute:: default_item_class
|
||||
|
||||
An Item class (or factory), used to instantiate items when not given in
|
||||
the constructor.
|
||||
the ``__init__`` method.
|
||||
|
||||
.. attribute:: default_input_processor
|
||||
|
||||
|
|
@ -515,15 +501,15 @@ ItemLoader objects
|
|||
.. attribute:: default_selector_class
|
||||
|
||||
The class used to construct the :attr:`selector` of this
|
||||
:class:`ItemLoader`, if only a response is given in the constructor.
|
||||
If a selector is given in the constructor this attribute is ignored.
|
||||
:class:`ItemLoader`, if only a response is given in the ``__init__`` method.
|
||||
If a selector is given in the ``__init__`` method this attribute is ignored.
|
||||
This attribute is sometimes overridden in subclasses.
|
||||
|
||||
.. attribute:: selector
|
||||
|
||||
The :class:`~scrapy.selector.Selector` object to extract data from.
|
||||
It's either the selector given in the constructor or one created from
|
||||
the response given in the constructor using the
|
||||
It's either the selector given in the ``__init__`` method or one created from
|
||||
the response given in the ``__init__`` method using the
|
||||
:attr:`default_selector_class`. This attribute is meant to be
|
||||
read-only.
|
||||
|
||||
|
|
@ -648,46 +634,46 @@ Here is a list of all built-in processors:
|
|||
.. class:: Identity
|
||||
|
||||
The simplest processor, which doesn't do anything. It returns the original
|
||||
values unchanged. It doesn't receive any constructor arguments, nor does it
|
||||
values unchanged. It doesn't receive any ``__init__`` method arguments, nor does it
|
||||
accept Loader contexts.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import Identity
|
||||
>>> proc = Identity()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
['one', 'two', 'three']
|
||||
>>> from scrapy.loader.processors import Identity
|
||||
>>> proc = Identity()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
['one', 'two', 'three']
|
||||
|
||||
.. class:: TakeFirst
|
||||
|
||||
Returns the first non-null/non-empty value from the values received,
|
||||
so it's typically used as an output processor to single-valued fields.
|
||||
It doesn't receive any constructor arguments, nor does it accept Loader contexts.
|
||||
It doesn't receive any ``__init__`` method arguments, nor does it accept Loader contexts.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> proc = TakeFirst()
|
||||
>>> proc(['', 'one', 'two', 'three'])
|
||||
'one'
|
||||
>>> from scrapy.loader.processors import TakeFirst
|
||||
>>> proc = TakeFirst()
|
||||
>>> proc(['', 'one', 'two', 'three'])
|
||||
'one'
|
||||
|
||||
.. class:: Join(separator=u' ')
|
||||
|
||||
Returns the values joined with the separator given in the constructor, which
|
||||
Returns the values joined with the separator given in the ``__init__`` method, which
|
||||
defaults to ``u' '``. It doesn't accept Loader contexts.
|
||||
|
||||
When using the default separator, this processor is equivalent to the
|
||||
function: ``u' '.join``
|
||||
|
||||
Examples::
|
||||
Examples:
|
||||
|
||||
>>> from scrapy.loader.processors import Join
|
||||
>>> proc = Join()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one two three'
|
||||
>>> proc = Join('<br>')
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one<br>two<br>three'
|
||||
>>> from scrapy.loader.processors import Join
|
||||
>>> proc = Join()
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one two three'
|
||||
>>> proc = Join('<br>')
|
||||
>>> proc(['one', 'two', 'three'])
|
||||
'one<br>two<br>three'
|
||||
|
||||
.. class:: Compose(\*functions, \**default_loader_context)
|
||||
|
||||
|
|
@ -700,18 +686,18 @@ Here is a list of all built-in processors:
|
|||
By default, stop process on ``None`` value. This behaviour can be changed by
|
||||
passing keyword argument ``stop_on_none=False``.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import Compose
|
||||
>>> proc = Compose(lambda v: v[0], str.upper)
|
||||
>>> proc(['hello', 'world'])
|
||||
'HELLO'
|
||||
>>> from scrapy.loader.processors import Compose
|
||||
>>> proc = Compose(lambda v: v[0], str.upper)
|
||||
>>> proc(['hello', 'world'])
|
||||
'HELLO'
|
||||
|
||||
Each function can optionally receive a ``loader_context`` parameter. For
|
||||
those which do, this processor will pass the currently active :ref:`Loader
|
||||
context <topics-loaders-context>` through that parameter.
|
||||
|
||||
The keyword arguments passed in the constructor are used as the default
|
||||
The keyword arguments passed in the ``__init__`` method are used as the default
|
||||
Loader context values passed to each function call. However, the final
|
||||
Loader context values passed to functions are overridden with the currently
|
||||
active Loader context accessible through the :meth:`ItemLoader.context`
|
||||
|
|
@ -744,41 +730,41 @@ Here is a list of all built-in processors:
|
|||
:meth:`~scrapy.selector.Selector.extract` method of :ref:`selectors
|
||||
<topics-selectors>`, which returns a list of unicode strings.
|
||||
|
||||
The example below should clarify how it works::
|
||||
The example below should clarify how it works:
|
||||
|
||||
>>> def filter_world(x):
|
||||
... return None if x == 'world' else x
|
||||
...
|
||||
>>> from scrapy.loader.processors import MapCompose
|
||||
>>> proc = MapCompose(filter_world, str.upper)
|
||||
>>> proc(['hello', 'world', 'this', 'is', 'scrapy'])
|
||||
['HELLO, 'THIS', 'IS', 'SCRAPY']
|
||||
>>> def filter_world(x):
|
||||
... return None if x == 'world' else x
|
||||
...
|
||||
>>> from scrapy.loader.processors import MapCompose
|
||||
>>> proc = MapCompose(filter_world, str.upper)
|
||||
>>> proc(['hello', 'world', 'this', 'is', 'scrapy'])
|
||||
['HELLO, 'THIS', 'IS', 'SCRAPY']
|
||||
|
||||
As with the Compose processor, functions can receive Loader contexts, and
|
||||
constructor keyword arguments are used as default context values. See
|
||||
``__init__`` method keyword arguments are used as default context values. See
|
||||
:class:`Compose` processor for more info.
|
||||
|
||||
.. class:: SelectJmes(json_path)
|
||||
|
||||
Queries the value using the json path provided to the constructor and returns the output.
|
||||
Queries the value using the json path provided to the ``__init__`` method and returns the output.
|
||||
Requires jmespath (https://github.com/jmespath/jmespath.py) to run.
|
||||
This processor takes only one input at a time.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose
|
||||
>>> proc = SelectJmes("foo") #for direct use on lists and dictionaries
|
||||
>>> proc({'foo': 'bar'})
|
||||
'bar'
|
||||
>>> proc({'foo': {'bar': 'baz'}})
|
||||
{'bar': 'baz'}
|
||||
>>> from scrapy.loader.processors import SelectJmes, Compose, MapCompose
|
||||
>>> proc = SelectJmes("foo") #for direct use on lists and dictionaries
|
||||
>>> proc({'foo': 'bar'})
|
||||
'bar'
|
||||
>>> proc({'foo': {'bar': 'baz'}})
|
||||
{'bar': 'baz'}
|
||||
|
||||
Working with Json::
|
||||
Working with Json:
|
||||
|
||||
>>> import json
|
||||
>>> proc_single_json_str = Compose(json.loads, SelectJmes("foo"))
|
||||
>>> proc_single_json_str('{"foo": "bar"}')
|
||||
'bar'
|
||||
>>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo')))
|
||||
>>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]')
|
||||
['bar']
|
||||
>>> import json
|
||||
>>> proc_single_json_str = Compose(json.loads, SelectJmes("foo"))
|
||||
>>> proc_single_json_str('{"foo": "bar"}')
|
||||
'bar'
|
||||
>>> proc_json_list = Compose(json.loads, MapCompose(SelectJmes('foo')))
|
||||
>>> proc_json_list('[{"foo":"bar"}, {"baz":"tar"}]')
|
||||
['bar']
|
||||
|
|
|
|||
|
|
@ -171,9 +171,9 @@ listed in `logging's logrecord attributes docs
|
|||
<https://docs.python.org/2/library/datetime.html#strftime-and-strptime-behavior>`_
|
||||
respectively.
|
||||
|
||||
If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the scrapy
|
||||
If :setting:`LOG_SHORT_NAMES` is set, then the logs will not display the Scrapy
|
||||
component that prints the log. It is unset by default, hence logs contain the
|
||||
scrapy component responsible for that log output.
|
||||
Scrapy component responsible for that log output.
|
||||
|
||||
Command-line options
|
||||
--------------------
|
||||
|
|
@ -255,18 +255,18 @@ scrapy.utils.log module
|
|||
when running custom scripts using :class:`~scrapy.crawler.CrawlerRunner`.
|
||||
In that case, its usage is not required but it's recommended.
|
||||
|
||||
If you plan on configuring the handlers yourself is still recommended you
|
||||
call this function, passing ``install_root_handler=False``. Bear in mind
|
||||
there won't be any log output set by default in that case.
|
||||
Another option when running custom scripts is to manually configure the logging.
|
||||
To do this you can use `logging.basicConfig()`_ to set a basic root handler.
|
||||
|
||||
To get you started on manually configuring logging's output, you can use
|
||||
`logging.basicConfig()`_ to set a basic root handler. This is an example
|
||||
on how to redirect ``INFO`` or higher messages to a file::
|
||||
Note that :class:`~scrapy.crawler.CrawlerProcess` automatically calls ``configure_logging``,
|
||||
so it is recommended to only use `logging.basicConfig()`_ together with
|
||||
:class:`~scrapy.crawler.CrawlerRunner`.
|
||||
|
||||
This is an example on how to redirect ``INFO`` or higher messages to a file::
|
||||
|
||||
import logging
|
||||
from scrapy.utils.log import configure_logging
|
||||
|
||||
configure_logging(install_root_handler=False)
|
||||
logging.basicConfig(
|
||||
filename='log.txt',
|
||||
format='%(levelname)s: %(message)s',
|
||||
|
|
|
|||
|
|
@ -97,7 +97,6 @@ For Files Pipeline, use::
|
|||
|
||||
ITEM_PIPELINES = {'scrapy.pipelines.files.FilesPipeline': 1}
|
||||
|
||||
|
||||
.. note::
|
||||
You can also use both the Files and Images Pipeline at the same time.
|
||||
|
||||
|
|
@ -148,6 +147,25 @@ Where:
|
|||
* ``full`` is a sub-directory to separate full images from thumbnails (if
|
||||
used). For more info see :ref:`topics-images-thumbnails`.
|
||||
|
||||
FTP server storage
|
||||
------------------
|
||||
|
||||
:setting:`FILES_STORE` and :setting:`IMAGES_STORE` can point to an FTP server.
|
||||
Scrapy will automatically upload the files to the server.
|
||||
|
||||
:setting:`FILES_STORE` and :setting:`IMAGES_STORE` should be written in one of the
|
||||
following forms::
|
||||
|
||||
ftp://username:password@address:port/path
|
||||
ftp://address:port/path
|
||||
|
||||
If ``username`` and ``password`` are not provided, they are taken from the :setting:`FTP_USER` and
|
||||
:setting:`FTP_PASSWORD` settings respectively.
|
||||
|
||||
FTP supports two different connection modes: active or passive. Scrapy uses
|
||||
the passive connection mode by default. To use the active connection mode instead,
|
||||
set the :setting:`FEED_STORAGE_FTP_ACTIVE` setting to ``True``.
|
||||
|
||||
Amazon S3 storage
|
||||
-----------------
|
||||
|
||||
|
|
@ -171,7 +189,7 @@ policy::
|
|||
|
||||
For more information, see `canned ACLs`_ in the Amazon S3 Developer Guide.
|
||||
|
||||
Because Scrapy uses ``boto`` / ``botocore`` internally you can also use other S3-like storages. Storages like
|
||||
Because Scrapy uses ``botocore`` internally you can also use other S3-like storages. Storages like
|
||||
self-hosted `Minio`_ or `s3.scality`_. All you need to do is set endpoint option in you Scrapy settings::
|
||||
|
||||
AWS_ENDPOINT_URL = 'http://minio.example.com:9000'
|
||||
|
|
@ -392,7 +410,7 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
|
||||
.. class:: FilesPipeline
|
||||
|
||||
.. method:: file_path(request, response, info)
|
||||
.. method:: file_path(self, request, response=None, info=None)
|
||||
|
||||
This method is called once per downloaded item. It returns the
|
||||
download path of the file originating from the specified
|
||||
|
|
@ -416,7 +434,7 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
|
||||
class MyFilesPipeline(FilesPipeline):
|
||||
|
||||
def file_path(self, request, response, info):
|
||||
def file_path(self, request, response=None, info=None):
|
||||
return 'files/' + os.path.basename(urlparse(request.url).path)
|
||||
|
||||
By default the :meth:`file_path` method returns
|
||||
|
|
@ -441,8 +459,9 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
* ``success`` is a boolean which is ``True`` if the image was downloaded
|
||||
successfully or ``False`` if it failed for some reason
|
||||
|
||||
* ``file_info_or_error`` is a dict containing the following keys (if success
|
||||
is ``True``) or a `Twisted Failure`_ if there was a problem.
|
||||
* ``file_info_or_error`` is a dict containing the following keys (if
|
||||
success is ``True``) or a :exc:`~twisted.python.failure.Failure` if
|
||||
there was a problem.
|
||||
|
||||
* ``url`` - the url where the file was downloaded from. This is the url of
|
||||
the request returned from the :meth:`~get_media_requests`
|
||||
|
|
@ -505,7 +524,7 @@ See here the methods that you can override in your custom Images Pipeline:
|
|||
The :class:`ImagesPipeline` is an extension of the :class:`FilesPipeline`,
|
||||
customizing the field names and adding custom behavior for images.
|
||||
|
||||
.. method:: file_path(request, response, info)
|
||||
.. method:: file_path(self, request, response=None, info=None)
|
||||
|
||||
This method is called once per downloaded item. It returns the
|
||||
download path of the file originating from the specified
|
||||
|
|
@ -529,7 +548,7 @@ See here the methods that you can override in your custom Images Pipeline:
|
|||
|
||||
class MyImagesPipeline(ImagesPipeline):
|
||||
|
||||
def file_path(self, request, response, info):
|
||||
def file_path(self, request, response=None, info=None):
|
||||
return 'files/' + os.path.basename(urlparse(request.url).path)
|
||||
|
||||
By default the :meth:`file_path` method returns
|
||||
|
|
@ -557,7 +576,7 @@ See here the methods that you can override in your custom Images Pipeline:
|
|||
Custom Images pipeline example
|
||||
==============================
|
||||
|
||||
Here is a full example of the Images Pipeline whose methods are examplified
|
||||
Here is a full example of the Images Pipeline whose methods are exemplified
|
||||
above::
|
||||
|
||||
import scrapy
|
||||
|
|
@ -577,5 +596,12 @@ above::
|
|||
item['image_paths'] = image_paths
|
||||
return item
|
||||
|
||||
.. _Twisted Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html
|
||||
|
||||
To enable your custom media pipeline component you must add its class import path to the
|
||||
:setting:`ITEM_PIPELINES` setting, like in the following example::
|
||||
|
||||
ITEM_PIPELINES = {
|
||||
'myproject.pipelines.MyImagesPipeline': 300
|
||||
}
|
||||
|
||||
.. _MD5 hash: https://en.wikipedia.org/wiki/MD5
|
||||
|
|
|
|||
|
|
@ -101,7 +101,7 @@ reactor after ``MySpider`` has finished running.
|
|||
d.addBoth(lambda _: reactor.stop())
|
||||
reactor.run() # the script will block here until the crawling is finished
|
||||
|
||||
.. seealso:: `Twisted Reactor Overview`_.
|
||||
.. seealso:: :doc:`twisted:core/howto/reactor-basics`
|
||||
|
||||
.. _run-multiple-spiders:
|
||||
|
||||
|
|
@ -253,6 +253,5 @@ If you are still unable to prevent your bot getting banned, consider contacting
|
|||
.. _ProxyMesh: https://proxymesh.com/
|
||||
.. _Google cache: http://www.googleguide.com/cached_pages.html
|
||||
.. _testspiders: https://github.com/scrapinghub/testspiders
|
||||
.. _Twisted Reactor Overview: https://twistedmatrix.com/documents/current/core/howto/reactor-basics.html
|
||||
.. _Crawlera: https://scrapinghub.com/crawlera
|
||||
.. _scrapoxy: https://scrapoxy.io/
|
||||
|
|
|
|||
|
|
@ -121,8 +121,8 @@ Request objects
|
|||
|
||||
:param errback: a function that will be called if any exception was
|
||||
raised while processing the request. This includes pages that failed
|
||||
with 404 HTTP errors and such. It receives a `Twisted Failure`_ instance
|
||||
as first parameter.
|
||||
with 404 HTTP errors and such. It receives a
|
||||
:exc:`~twisted.python.failure.Failure` as first parameter.
|
||||
For more information,
|
||||
see :ref:`topics-request-response-ref-errbacks` below.
|
||||
:type errback: callable
|
||||
|
|
@ -137,7 +137,7 @@ Request objects
|
|||
|
||||
A string containing the URL of this request. Keep in mind that this
|
||||
attribute contains the escaped URL, so it can differ from the URL passed in
|
||||
the constructor.
|
||||
the ``__init__`` method.
|
||||
|
||||
This attribute is read-only. To change the URL of a Request use
|
||||
:meth:`replace`.
|
||||
|
|
@ -254,8 +254,8 @@ Using errbacks to catch exceptions in request processing
|
|||
The errback of a request is a function that will be called when an exception
|
||||
is raise while processing it.
|
||||
|
||||
It receives a `Twisted Failure`_ instance as first parameter and can be
|
||||
used to track connection establishment timeouts, DNS errors etc.
|
||||
It receives a :exc:`~twisted.python.failure.Failure` as first parameter and can
|
||||
be used to track connection establishment timeouts, DNS errors etc.
|
||||
|
||||
Here's an example spider logging all errors and catching some specific
|
||||
errors if needed::
|
||||
|
|
@ -396,11 +396,11 @@ The FormRequest class extends the base :class:`Request` with functionality for
|
|||
dealing with HTML forms. It uses `lxml.html forms`_ to pre-populate form
|
||||
fields with form data from :class:`Response` objects.
|
||||
|
||||
.. _lxml.html forms: http://lxml.de/lxmlhtml.html#forms
|
||||
.. _lxml.html forms: https://lxml.de/lxmlhtml.html#forms
|
||||
|
||||
.. class:: FormRequest(url, [formdata, ...])
|
||||
|
||||
The :class:`FormRequest` class adds a new keyword parameter to the constructor. The
|
||||
The :class:`FormRequest` class adds a new keyword parameter to the ``__init__`` method. The
|
||||
remaining arguments are the same as for the :class:`Request` class and are
|
||||
not documented here.
|
||||
|
||||
|
|
@ -473,7 +473,7 @@ fields with form data from :class:`Response` objects.
|
|||
:type dont_click: boolean
|
||||
|
||||
The other parameters of this class method are passed directly to the
|
||||
:class:`FormRequest` constructor.
|
||||
:class:`FormRequest` ``__init__`` method.
|
||||
|
||||
.. versionadded:: 0.10.3
|
||||
The ``formname`` parameter.
|
||||
|
|
@ -547,7 +547,7 @@ dealing with JSON requests.
|
|||
|
||||
.. class:: JsonRequest(url, [... data, dumps_kwargs])
|
||||
|
||||
The :class:`JsonRequest` class adds two new keyword parameters to the constructor. The
|
||||
The :class:`JsonRequest` class adds two new keyword parameters to the ``__init__`` method. The
|
||||
remaining arguments are the same as for the :class:`Request` class and are
|
||||
not documented here.
|
||||
|
||||
|
|
@ -556,7 +556,7 @@ dealing with JSON requests.
|
|||
|
||||
:param data: is any JSON serializable object that needs to be JSON encoded and assigned to body.
|
||||
if :attr:`Request.body` argument is provided this parameter will be ignored.
|
||||
if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be
|
||||
if :attr:`Request.body` argument is not provided and data argument is provided :attr:`Request.method` will be
|
||||
set to ``'POST'`` automatically.
|
||||
:type data: JSON serializable object
|
||||
|
||||
|
|
@ -596,8 +596,8 @@ Response objects
|
|||
(for single valued headers) or lists (for multi-valued headers).
|
||||
:type headers: dict
|
||||
|
||||
:param body: the response body. To access the decoded text as str (unicode
|
||||
in Python 2) you can use ``response.text`` from an encoding-aware
|
||||
:param body: the response body. To access the decoded text as str you can use
|
||||
``response.text`` from an encoding-aware
|
||||
:ref:`Response subclass <topics-request-response-ref-response-subclasses>`,
|
||||
such as :class:`TextResponse`.
|
||||
:type body: bytes
|
||||
|
|
@ -609,7 +609,10 @@ Response objects
|
|||
|
||||
:param request: the initial value of the :attr:`Response.request` attribute.
|
||||
This represents the :class:`Request` that generated this response.
|
||||
:type request: :class:`Request` object
|
||||
:type request: scrapy.http.Request
|
||||
|
||||
:param certificate: an object representing the server's SSL certificate.
|
||||
:type certificate: twisted.internet.ssl.Certificate
|
||||
|
||||
.. attribute:: Response.url
|
||||
|
||||
|
|
@ -664,7 +667,7 @@ Response objects
|
|||
.. attribute:: Response.meta
|
||||
|
||||
A shortcut to the :attr:`Request.meta` attribute of the
|
||||
:attr:`Response.request` object (ie. ``self.request.meta``).
|
||||
:attr:`Response.request` object (i.e. ``self.request.meta``).
|
||||
|
||||
Unlike the :attr:`Response.request` attribute, the :attr:`Response.meta`
|
||||
attribute is propagated along redirects and retries, so you will get
|
||||
|
|
@ -672,6 +675,18 @@ Response objects
|
|||
|
||||
.. seealso:: :attr:`Request.meta` attribute
|
||||
|
||||
.. attribute:: Response.cb_kwargs
|
||||
|
||||
A shortcut to the :attr:`Request.cb_kwargs` attribute of the
|
||||
:attr:`Response.request` object (i.e. ``self.request.cb_kwargs``).
|
||||
|
||||
Unlike the :attr:`Response.request` attribute, the
|
||||
:attr:`Response.cb_kwargs` attribute is propagated along redirects and
|
||||
retries, so you will get the original :attr:`Request.cb_kwargs` sent
|
||||
from your spider.
|
||||
|
||||
.. seealso:: :attr:`Request.cb_kwargs` attribute
|
||||
|
||||
.. attribute:: Response.flags
|
||||
|
||||
A list that contains flags for this response. Flags are labels used for
|
||||
|
|
@ -679,6 +694,13 @@ Response objects
|
|||
they're shown on the string representation of the Response (`__str__`
|
||||
method) which is used by the engine for logging.
|
||||
|
||||
.. attribute:: Response.certificate
|
||||
|
||||
A :class:`twisted.internet.ssl.Certificate` object representing
|
||||
the server's SSL certificate.
|
||||
|
||||
Only populated for ``https`` responses, ``None`` otherwise.
|
||||
|
||||
.. method:: Response.copy()
|
||||
|
||||
Returns a new Response which is a copy of this Response.
|
||||
|
|
@ -701,6 +723,8 @@ Response objects
|
|||
|
||||
.. automethod:: Response.follow
|
||||
|
||||
.. automethod:: Response.follow_all
|
||||
|
||||
|
||||
.. _urlparse.urljoin: https://docs.python.org/2/library/urlparse.html#urlparse.urljoin
|
||||
|
||||
|
|
@ -721,7 +745,7 @@ TextResponse objects
|
|||
:class:`Response` class, which is meant to be used only for binary data,
|
||||
such as images, sounds or any media file.
|
||||
|
||||
:class:`TextResponse` objects support a new constructor argument, in
|
||||
:class:`TextResponse` objects support a new ``__init__`` method argument, in
|
||||
addition to the base :class:`Response` objects. The remaining functionality
|
||||
is the same as for the :class:`Response` class and is not documented here.
|
||||
|
||||
|
|
@ -755,10 +779,10 @@ TextResponse objects
|
|||
A string with the encoding of this response. The encoding is resolved by
|
||||
trying the following mechanisms, in order:
|
||||
|
||||
1. the encoding passed in the constructor ``encoding`` argument
|
||||
1. the encoding passed in the ``__init__`` method ``encoding`` argument
|
||||
|
||||
2. the encoding declared in the Content-Type HTTP header. If this
|
||||
encoding is not valid (ie. unknown), it is ignored and the next
|
||||
encoding is not valid (i.e. unknown), it is ignored and the next
|
||||
resolution mechanism is tried.
|
||||
|
||||
3. the encoding declared in the response body. The TextResponse class
|
||||
|
|
@ -790,6 +814,8 @@ TextResponse objects
|
|||
|
||||
.. automethod:: TextResponse.follow
|
||||
|
||||
.. automethod:: TextResponse.follow_all
|
||||
|
||||
.. method:: TextResponse.body_as_unicode()
|
||||
|
||||
The same as :attr:`text`, but available as a method. This method is
|
||||
|
|
@ -816,5 +842,4 @@ XmlResponse objects
|
|||
adds encoding auto-discovering support by looking into the XML declaration
|
||||
line. See :attr:`TextResponse.encoding`.
|
||||
|
||||
.. _Twisted Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html
|
||||
.. _bug in lxml: https://bugs.launchpad.net/lxml/+bug/1665241
|
||||
|
|
|
|||
|
|
@ -35,12 +35,11 @@ defines selectors to associate those styles with specific HTML elements.
|
|||
in speed and parsing accuracy to lxml.
|
||||
|
||||
.. _BeautifulSoup: https://www.crummy.com/software/BeautifulSoup/
|
||||
.. _lxml: http://lxml.de/
|
||||
.. _lxml: https://lxml.de/
|
||||
.. _ElementTree: https://docs.python.org/2/library/xml.etree.elementtree.html
|
||||
.. _cssselect: https://pypi.python.org/pypi/cssselect/
|
||||
.. _XPath: https://www.w3.org/TR/xpath
|
||||
.. _XPath: https://www.w3.org/TR/xpath/all/
|
||||
.. _CSS: https://www.w3.org/TR/selectors
|
||||
.. _parsel: https://parsel.readthedocs.io/
|
||||
.. _parsel: https://parsel.readthedocs.io/en/latest/
|
||||
|
||||
Using selectors
|
||||
===============
|
||||
|
|
@ -51,18 +50,18 @@ Constructing selectors
|
|||
.. highlight:: python
|
||||
|
||||
Response objects expose a :class:`~scrapy.selector.Selector` instance
|
||||
on ``.selector`` attribute::
|
||||
on ``.selector`` attribute:
|
||||
|
||||
>>> response.selector.xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> response.selector.xpath('//span/text()').get()
|
||||
'good'
|
||||
|
||||
Querying responses using XPath and CSS is so common that responses include two
|
||||
more shortcuts: ``response.xpath()`` and ``response.css()``::
|
||||
more shortcuts: ``response.xpath()`` and ``response.css()``:
|
||||
|
||||
>>> response.xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> response.css('span::text').get()
|
||||
'good'
|
||||
>>> response.xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> response.css('span::text').get()
|
||||
'good'
|
||||
|
||||
Scrapy selectors are instances of :class:`~scrapy.selector.Selector` class
|
||||
constructed by passing either :class:`~scrapy.http.TextResponse` object or
|
||||
|
|
@ -74,21 +73,21 @@ shortcuts. By using ``response.selector`` or one of these shortcuts
|
|||
you can also ensure the response body is parsed only once.
|
||||
|
||||
But if required, it is possible to use ``Selector`` directly.
|
||||
Constructing from text::
|
||||
Constructing from text:
|
||||
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> body = '<html><body><span>good</span></body></html>'
|
||||
>>> Selector(text=body).xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> body = '<html><body><span>good</span></body></html>'
|
||||
>>> Selector(text=body).xpath('//span/text()').get()
|
||||
'good'
|
||||
|
||||
Constructing from response - :class:`~scrapy.http.HtmlResponse` is one of
|
||||
:class:`~scrapy.http.TextResponse` subclasses::
|
||||
:class:`~scrapy.http.TextResponse` subclasses:
|
||||
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> from scrapy.http import HtmlResponse
|
||||
>>> response = HtmlResponse(url='http://example.com', body=body)
|
||||
>>> Selector(response=response).xpath('//span/text()').get()
|
||||
'good'
|
||||
>>> from scrapy.selector import Selector
|
||||
>>> from scrapy.http import HtmlResponse
|
||||
>>> response = HtmlResponse(url='http://example.com', body=body)
|
||||
>>> Selector(response=response).xpath('//span/text()').get()
|
||||
'good'
|
||||
|
||||
``Selector`` automatically chooses the best parsing rules
|
||||
(XML vs HTML) based on input type.
|
||||
|
|
@ -123,118 +122,118 @@ Since we're dealing with HTML, the selector will automatically use an HTML parse
|
|||
.. highlight:: python
|
||||
|
||||
So, by looking at the :ref:`HTML code <topics-selectors-htmlcode>` of that
|
||||
page, let's construct an XPath for selecting the text inside the title tag::
|
||||
page, let's construct an XPath for selecting the text inside the title tag:
|
||||
|
||||
>>> response.xpath('//title/text()')
|
||||
[<Selector xpath='//title/text()' data='Example website'>]
|
||||
>>> response.xpath('//title/text()')
|
||||
[<Selector xpath='//title/text()' data='Example website'>]
|
||||
|
||||
To actually extract the textual data, you must call the selector ``.get()``
|
||||
or ``.getall()`` methods, as follows::
|
||||
or ``.getall()`` methods, as follows:
|
||||
|
||||
>>> response.xpath('//title/text()').getall()
|
||||
['Example website']
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Example website'
|
||||
>>> response.xpath('//title/text()').getall()
|
||||
['Example website']
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Example website'
|
||||
|
||||
``.get()`` always returns a single result; if there are several matches,
|
||||
content of a first match is returned; if there are no matches, None
|
||||
is returned. ``.getall()`` returns a list with all results.
|
||||
|
||||
Notice that CSS selectors can select text or attribute nodes using CSS3
|
||||
pseudo-elements::
|
||||
pseudo-elements:
|
||||
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
|
||||
As you can see, ``.xpath()`` and ``.css()`` methods return a
|
||||
:class:`~scrapy.selector.SelectorList` instance, which is a list of new
|
||||
selectors. This API can be used for quickly selecting nested data::
|
||||
selectors. This API can be used for quickly selecting nested data:
|
||||
|
||||
>>> response.css('img').xpath('@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> response.css('img').xpath('@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
If you want to extract only the first matched element, you can call the
|
||||
selector ``.get()`` (or its alias ``.extract_first()`` commonly used in
|
||||
previous Scrapy versions)::
|
||||
previous Scrapy versions):
|
||||
|
||||
>>> response.xpath('//div[@id="images"]/a/text()').get()
|
||||
'Name: My image 1 '
|
||||
>>> response.xpath('//div[@id="images"]/a/text()').get()
|
||||
'Name: My image 1 '
|
||||
|
||||
It returns ``None`` if no element was found::
|
||||
It returns ``None`` if no element was found:
|
||||
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get() is None
|
||||
True
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get() is None
|
||||
True
|
||||
|
||||
A default return value can be provided as an argument, to be used instead
|
||||
of ``None``:
|
||||
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found')
|
||||
'not-found'
|
||||
>>> response.xpath('//div[@id="not-exists"]/text()').get(default='not-found')
|
||||
'not-found'
|
||||
|
||||
Instead of using e.g. ``'@src'`` XPath it is possible to query for attributes
|
||||
using ``.attrib`` property of a :class:`~scrapy.selector.Selector`::
|
||||
using ``.attrib`` property of a :class:`~scrapy.selector.Selector`:
|
||||
|
||||
>>> [img.attrib['src'] for img in response.css('img')]
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> [img.attrib['src'] for img in response.css('img')]
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
As a shortcut, ``.attrib`` is also available on SelectorList directly;
|
||||
it returns attributes for the first matching element::
|
||||
it returns attributes for the first matching element:
|
||||
|
||||
>>> response.css('img').attrib['src']
|
||||
'image1_thumb.jpg'
|
||||
>>> response.css('img').attrib['src']
|
||||
'image1_thumb.jpg'
|
||||
|
||||
This is most useful when only a single result is expected, e.g. when selecting
|
||||
by id, or selecting unique elements on a web page::
|
||||
by id, or selecting unique elements on a web page:
|
||||
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
|
||||
Now we're going to get the base URL and some image links::
|
||||
Now we're going to get the base URL and some image links:
|
||||
|
||||
>>> response.xpath('//base/@href').get()
|
||||
'http://example.com/'
|
||||
>>> response.xpath('//base/@href').get()
|
||||
'http://example.com/'
|
||||
|
||||
>>> response.css('base::attr(href)').get()
|
||||
'http://example.com/'
|
||||
>>> response.css('base::attr(href)').get()
|
||||
'http://example.com/'
|
||||
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/@href').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
>>> response.xpath('//a[contains(@href, "image")]/@href').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
|
||||
>>> response.css('a[href*=image]::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
>>> response.css('a[href*=image]::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/img/@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> response.xpath('//a[contains(@href, "image")]/img/@src').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
>>> response.css('a[href*=image] img::attr(src)').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
>>> response.css('a[href*=image] img::attr(src)').getall()
|
||||
['image1_thumb.jpg',
|
||||
'image2_thumb.jpg',
|
||||
'image3_thumb.jpg',
|
||||
'image4_thumb.jpg',
|
||||
'image5_thumb.jpg']
|
||||
|
||||
.. _topics-selectors-css-extensions:
|
||||
|
||||
|
|
@ -255,51 +254,51 @@ that Scrapy (parsel) implements a couple of **non-standard pseudo-elements**:
|
|||
They will most probably not work with other libraries like
|
||||
`lxml`_ or `PyQuery`_.
|
||||
|
||||
.. _PyQuery: https://pypi.python.org/pypi/pyquery
|
||||
.. _PyQuery: https://pypi.org/project/pyquery/
|
||||
|
||||
Examples:
|
||||
|
||||
* ``title::text`` selects children text nodes of a descendant ``<title>`` element::
|
||||
* ``title::text`` selects children text nodes of a descendant ``<title>`` element:
|
||||
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
>>> response.css('title::text').get()
|
||||
'Example website'
|
||||
|
||||
* ``*::text`` selects all descendant text nodes of the current selector context::
|
||||
* ``*::text`` selects all descendant text nodes of the current selector context:
|
||||
|
||||
>>> response.css('#images *::text').getall()
|
||||
['\n ',
|
||||
'Name: My image 1 ',
|
||||
'\n ',
|
||||
'Name: My image 2 ',
|
||||
'\n ',
|
||||
'Name: My image 3 ',
|
||||
'\n ',
|
||||
'Name: My image 4 ',
|
||||
'\n ',
|
||||
'Name: My image 5 ',
|
||||
'\n ']
|
||||
>>> response.css('#images *::text').getall()
|
||||
['\n ',
|
||||
'Name: My image 1 ',
|
||||
'\n ',
|
||||
'Name: My image 2 ',
|
||||
'\n ',
|
||||
'Name: My image 3 ',
|
||||
'\n ',
|
||||
'Name: My image 4 ',
|
||||
'\n ',
|
||||
'Name: My image 5 ',
|
||||
'\n ']
|
||||
|
||||
* ``foo::text`` returns no results if ``foo`` element exists, but contains
|
||||
no text (i.e. text is empty)::
|
||||
no text (i.e. text is empty):
|
||||
|
||||
>>> response.css('img::text').getall()
|
||||
[]
|
||||
>>> response.css('img::text').getall()
|
||||
[]
|
||||
|
||||
This means ``.css('foo::text').get()`` could return None even if an element
|
||||
exists. Use ``default=''`` if you always want a string::
|
||||
exists. Use ``default=''`` if you always want a string:
|
||||
|
||||
>>> response.css('img::text').get()
|
||||
>>> response.css('img::text').get(default='')
|
||||
''
|
||||
>>> response.css('img::text').get()
|
||||
>>> response.css('img::text').get(default='')
|
||||
''
|
||||
|
||||
* ``a::attr(href)`` selects the *href* attribute value of descendant links::
|
||||
* ``a::attr(href)`` selects the *href* attribute value of descendant links:
|
||||
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html',
|
||||
'image2.html',
|
||||
'image3.html',
|
||||
'image4.html',
|
||||
'image5.html']
|
||||
|
||||
.. note::
|
||||
See also: :ref:`selecting-attributes`.
|
||||
|
|
@ -309,7 +308,7 @@ Examples:
|
|||
make much sense: text nodes do not have attributes, and attribute values
|
||||
are string values already and do not have children nodes.
|
||||
|
||||
.. _CSS Selectors: https://www.w3.org/TR/css3-selectors/#selectors
|
||||
.. _CSS Selectors: https://www.w3.org/TR/selectors-3/#selectors
|
||||
|
||||
.. _topics-selectors-nesting-selectors:
|
||||
|
||||
|
|
@ -318,25 +317,24 @@ Nesting selectors
|
|||
|
||||
The selection methods (``.xpath()`` or ``.css()``) return a list of selectors
|
||||
of the same type, so you can call the selection methods for those selectors
|
||||
too. Here's an example::
|
||||
too. Here's an example:
|
||||
|
||||
>>> links = response.xpath('//a[contains(@href, "image")]')
|
||||
>>> links.getall()
|
||||
['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>',
|
||||
'<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>',
|
||||
'<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>',
|
||||
'<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>',
|
||||
'<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>']
|
||||
>>> links = response.xpath('//a[contains(@href, "image")]')
|
||||
>>> links.getall()
|
||||
['<a href="image1.html">Name: My image 1 <br><img src="image1_thumb.jpg"></a>',
|
||||
'<a href="image2.html">Name: My image 2 <br><img src="image2_thumb.jpg"></a>',
|
||||
'<a href="image3.html">Name: My image 3 <br><img src="image3_thumb.jpg"></a>',
|
||||
'<a href="image4.html">Name: My image 4 <br><img src="image4_thumb.jpg"></a>',
|
||||
'<a href="image5.html">Name: My image 5 <br><img src="image5_thumb.jpg"></a>']
|
||||
|
||||
>>> for index, link in enumerate(links):
|
||||
... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get())
|
||||
... print('Link number %d points to url %r and image %r' % args)
|
||||
|
||||
Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg'
|
||||
Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg'
|
||||
Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg'
|
||||
Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg'
|
||||
Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg'
|
||||
>>> for index, link in enumerate(links):
|
||||
... args = (index, link.xpath('@href').get(), link.xpath('img/@src').get())
|
||||
... print('Link number %d points to url %r and image %r' % args)
|
||||
Link number 0 points to url 'image1.html' and image 'image1_thumb.jpg'
|
||||
Link number 1 points to url 'image2.html' and image 'image2_thumb.jpg'
|
||||
Link number 2 points to url 'image3.html' and image 'image3_thumb.jpg'
|
||||
Link number 3 points to url 'image4.html' and image 'image4_thumb.jpg'
|
||||
Link number 4 points to url 'image5.html' and image 'image5_thumb.jpg'
|
||||
|
||||
.. _selecting-attributes:
|
||||
|
||||
|
|
@ -344,42 +342,42 @@ Selecting element attributes
|
|||
----------------------------
|
||||
|
||||
There are several ways to get a value of an attribute. First, one can use
|
||||
XPath syntax::
|
||||
XPath syntax:
|
||||
|
||||
>>> response.xpath("//a/@href").getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.xpath("//a/@href").getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
XPath syntax has a few advantages: it is a standard XPath feature, and
|
||||
``@attributes`` can be used in other parts of an XPath expression - e.g.
|
||||
it is possible to filter by attribute value.
|
||||
|
||||
Scrapy also provides an extension to CSS selectors (``::attr(...)``)
|
||||
which allows to get attribute values::
|
||||
which allows to get attribute values:
|
||||
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
In addition to that, there is a ``.attrib`` property of Selector.
|
||||
You can use it if you prefer to lookup attributes in Python
|
||||
code, without using XPaths or CSS extensions::
|
||||
code, without using XPaths or CSS extensions:
|
||||
|
||||
>>> [a.attrib['href'] for a in response.css('a')]
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> [a.attrib['href'] for a in response.css('a')]
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
This property is also available on SelectorList; it returns a dictionary
|
||||
with attributes of a first matching element. It is convenient to use when
|
||||
a selector is expected to give a single result (e.g. when selecting by element
|
||||
ID, or when selecting an unique element on a page)::
|
||||
ID, or when selecting an unique element on a page):
|
||||
|
||||
>>> response.css('base').attrib
|
||||
{'href': 'http://example.com/'}
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
>>> response.css('base').attrib
|
||||
{'href': 'http://example.com/'}
|
||||
>>> response.css('base').attrib['href']
|
||||
'http://example.com/'
|
||||
|
||||
``.attrib`` property of an empty SelectorList is empty::
|
||||
``.attrib`` property of an empty SelectorList is empty:
|
||||
|
||||
>>> response.css('foo').attrib
|
||||
{}
|
||||
>>> response.css('foo').attrib
|
||||
{}
|
||||
|
||||
Using selectors with regular expressions
|
||||
----------------------------------------
|
||||
|
|
@ -390,21 +388,21 @@ data using regular expressions. However, unlike using ``.xpath()`` or
|
|||
can't construct nested ``.re()`` calls.
|
||||
|
||||
Here's an example used to extract image names from the :ref:`HTML code
|
||||
<topics-selectors-htmlcode>` above::
|
||||
<topics-selectors-htmlcode>` above:
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)')
|
||||
['My image 1',
|
||||
'My image 2',
|
||||
'My image 3',
|
||||
'My image 4',
|
||||
'My image 5']
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re(r'Name:\s*(.*)')
|
||||
['My image 1',
|
||||
'My image 2',
|
||||
'My image 3',
|
||||
'My image 4',
|
||||
'My image 5']
|
||||
|
||||
There's an additional helper reciprocating ``.get()`` (and its
|
||||
alias ``.extract_first()``) for ``.re()``, named ``.re_first()``.
|
||||
Use it to extract just the first matching string::
|
||||
Use it to extract just the first matching string:
|
||||
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)')
|
||||
'My image 1'
|
||||
>>> response.xpath('//a[contains(@href, "image")]/text()').re_first(r'Name:\s*(.*)')
|
||||
'My image 1'
|
||||
|
||||
.. _old-extraction-api:
|
||||
|
||||
|
|
@ -422,28 +420,28 @@ and readable code.
|
|||
|
||||
The following examples show how these methods map to each other.
|
||||
|
||||
1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``::
|
||||
1. ``SelectorList.get()`` is the same as ``SelectorList.extract_first()``:
|
||||
|
||||
>>> response.css('a::attr(href)').get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)').extract_first()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)').get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)').extract_first()
|
||||
'image1.html'
|
||||
|
||||
2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``::
|
||||
2. ``SelectorList.getall()`` is the same as ``SelectorList.extract()``:
|
||||
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').extract()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').getall()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
>>> response.css('a::attr(href)').extract()
|
||||
['image1.html', 'image2.html', 'image3.html', 'image4.html', 'image5.html']
|
||||
|
||||
3. ``Selector.get()`` is the same as ``Selector.extract()``::
|
||||
3. ``Selector.get()`` is the same as ``Selector.extract()``:
|
||||
|
||||
>>> response.css('a::attr(href)')[0].get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)')[0].extract()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)')[0].get()
|
||||
'image1.html'
|
||||
>>> response.css('a::attr(href)')[0].extract()
|
||||
'image1.html'
|
||||
|
||||
4. For consistency, there is also ``Selector.getall()``, which returns a list::
|
||||
4. For consistency, there is also ``Selector.getall()``, which returns a list:
|
||||
|
||||
>>> response.css('a::attr(href)')[0].getall()
|
||||
['image1.html']
|
||||
|
|
@ -481,31 +479,31 @@ with ``/``, that XPath will be absolute to the document and not relative to the
|
|||
``Selector`` you're calling it from.
|
||||
|
||||
For example, suppose you want to extract all ``<p>`` elements inside ``<div>``
|
||||
elements. First, you would get all ``<div>`` elements::
|
||||
elements. First, you would get all ``<div>`` elements:
|
||||
|
||||
>>> divs = response.xpath('//div')
|
||||
>>> divs = response.xpath('//div')
|
||||
|
||||
At first, you may be tempted to use the following approach, which is wrong, as
|
||||
it actually extracts all ``<p>`` elements from the document, not only those
|
||||
inside ``<div>`` elements::
|
||||
inside ``<div>`` elements:
|
||||
|
||||
>>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document
|
||||
... print(p.get())
|
||||
>>> for p in divs.xpath('//p'): # this is wrong - gets all <p> from the whole document
|
||||
... print(p.get())
|
||||
|
||||
This is the proper way to do it (note the dot prefixing the ``.//p`` XPath)::
|
||||
This is the proper way to do it (note the dot prefixing the ``.//p`` XPath):
|
||||
|
||||
>>> for p in divs.xpath('.//p'): # extracts all <p> inside
|
||||
... print(p.get())
|
||||
>>> for p in divs.xpath('.//p'): # extracts all <p> inside
|
||||
... print(p.get())
|
||||
|
||||
Another common case would be to extract all direct ``<p>`` children::
|
||||
Another common case would be to extract all direct ``<p>`` children:
|
||||
|
||||
>>> for p in divs.xpath('p'):
|
||||
... print(p.get())
|
||||
>>> for p in divs.xpath('p'):
|
||||
... print(p.get())
|
||||
|
||||
For more details about relative XPaths see the `Location Paths`_ section in the
|
||||
XPath specification.
|
||||
|
||||
.. _Location Paths: https://www.w3.org/TR/xpath#location-paths
|
||||
.. _Location Paths: https://www.w3.org/TR/xpath/all/#location-paths
|
||||
|
||||
When querying by class, consider using CSS
|
||||
------------------------------------------
|
||||
|
|
@ -521,12 +519,12 @@ for that you may end up with more elements that you want, if they have a differe
|
|||
class name that shares the string ``someclass``.
|
||||
|
||||
As it turns out, Scrapy selectors allow you to chain selectors, so most of the time
|
||||
you can just select by class using CSS and then switch to XPath when needed::
|
||||
you can just select by class using CSS and then switch to XPath when needed:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>')
|
||||
>>> sel.css('.shout').xpath('./time/@datetime').getall()
|
||||
['2014-07-23 19:00']
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>')
|
||||
>>> sel.css('.shout').xpath('./time/@datetime').getall()
|
||||
['2014-07-23 19:00']
|
||||
|
||||
This is cleaner than using the verbose XPath trick shown above. Just remember
|
||||
to use the ``.`` in the XPath expressions that will follow.
|
||||
|
|
@ -538,41 +536,41 @@ Beware of the difference between //node[1] and (//node)[1]
|
|||
|
||||
``(//node)[1]`` selects all the nodes in the document, and then gets only the first of them.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text="""
|
||||
....: <ul class="list">
|
||||
....: <li>1</li>
|
||||
....: <li>2</li>
|
||||
....: <li>3</li>
|
||||
....: </ul>
|
||||
....: <ul class="list">
|
||||
....: <li>4</li>
|
||||
....: <li>5</li>
|
||||
....: <li>6</li>
|
||||
....: </ul>""")
|
||||
>>> xp = lambda x: sel.xpath(x).getall()
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text="""
|
||||
....: <ul class="list">
|
||||
....: <li>1</li>
|
||||
....: <li>2</li>
|
||||
....: <li>3</li>
|
||||
....: </ul>
|
||||
....: <ul class="list">
|
||||
....: <li>4</li>
|
||||
....: <li>5</li>
|
||||
....: <li>6</li>
|
||||
....: </ul>""")
|
||||
>>> xp = lambda x: sel.xpath(x).getall()
|
||||
|
||||
This gets all first ``<li>`` elements under whatever it is its parent::
|
||||
This gets all first ``<li>`` elements under whatever it is its parent:
|
||||
|
||||
>>> xp("//li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
>>> xp("//li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
|
||||
And this gets the first ``<li>`` element in the whole document::
|
||||
And this gets the first ``<li>`` element in the whole document:
|
||||
|
||||
>>> xp("(//li)[1]")
|
||||
['<li>1</li>']
|
||||
>>> xp("(//li)[1]")
|
||||
['<li>1</li>']
|
||||
|
||||
This gets all first ``<li>`` elements under an ``<ul>`` parent::
|
||||
This gets all first ``<li>`` elements under an ``<ul>`` parent:
|
||||
|
||||
>>> xp("//ul/li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
>>> xp("//ul/li[1]")
|
||||
['<li>1</li>', '<li>4</li>']
|
||||
|
||||
And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document::
|
||||
And this gets the first ``<li>`` element under an ``<ul>`` parent in the whole document:
|
||||
|
||||
>>> xp("(//ul/li)[1]")
|
||||
['<li>1</li>']
|
||||
>>> xp("(//ul/li)[1]")
|
||||
['<li>1</li>']
|
||||
|
||||
Using text nodes in a condition
|
||||
-------------------------------
|
||||
|
|
@ -584,36 +582,36 @@ This is because the expression ``.//text()`` yields a collection of text element
|
|||
And when a node-set is converted to a string, which happens when it is passed as argument to
|
||||
a string function like ``contains()`` or ``starts-with()``, it results in the text for the first element only.
|
||||
|
||||
Example::
|
||||
Example:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>')
|
||||
>>> from scrapy import Selector
|
||||
>>> sel = Selector(text='<a href="#">Click here to go to the <strong>Next Page</strong></a>')
|
||||
|
||||
Converting a *node-set* to string::
|
||||
Converting a *node-set* to string:
|
||||
|
||||
>>> sel.xpath('//a//text()').getall() # take a peek at the node-set
|
||||
['Click here to go to the ', 'Next Page']
|
||||
>>> sel.xpath("string(//a[1]//text())").getall() # convert it to string
|
||||
['Click here to go to the ']
|
||||
>>> sel.xpath('//a//text()').getall() # take a peek at the node-set
|
||||
['Click here to go to the ', 'Next Page']
|
||||
>>> sel.xpath("string(//a[1]//text())").getall() # convert it to string
|
||||
['Click here to go to the ']
|
||||
|
||||
A *node* converted to a string, however, puts together the text of itself plus of all its descendants::
|
||||
A *node* converted to a string, however, puts together the text of itself plus of all its descendants:
|
||||
|
||||
>>> sel.xpath("//a[1]").getall() # select the first node
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
>>> sel.xpath("string(//a[1])").getall() # convert it to string
|
||||
['Click here to go to the Next Page']
|
||||
>>> sel.xpath("//a[1]").getall() # select the first node
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
>>> sel.xpath("string(//a[1])").getall() # convert it to string
|
||||
['Click here to go to the Next Page']
|
||||
|
||||
So, using the ``.//text()`` node-set won't select anything in this case::
|
||||
So, using the ``.//text()`` node-set won't select anything in this case:
|
||||
|
||||
>>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall()
|
||||
[]
|
||||
>>> sel.xpath("//a[contains(.//text(), 'Next Page')]").getall()
|
||||
[]
|
||||
|
||||
But using the ``.`` to mean the node, works::
|
||||
But using the ``.`` to mean the node, works:
|
||||
|
||||
>>> sel.xpath("//a[contains(., 'Next Page')]").getall()
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
>>> sel.xpath("//a[contains(., 'Next Page')]").getall()
|
||||
['<a href="#">Click here to go to the <strong>Next Page</strong></a>']
|
||||
|
||||
.. _`XPath string function`: https://www.w3.org/TR/xpath/#section-String-Functions
|
||||
.. _`XPath string function`: https://www.w3.org/TR/xpath/all/#section-String-Functions
|
||||
|
||||
.. _topics-selectors-xpath-variables:
|
||||
|
||||
|
|
@ -627,17 +625,17 @@ some arguments in your queries with placeholders like ``?``,
|
|||
which are then substituted with values passed with the query.
|
||||
|
||||
Here's an example to match an element based on its "id" attribute value,
|
||||
without hard-coding it (that was shown previously)::
|
||||
without hard-coding it (that was shown previously):
|
||||
|
||||
>>> # `$val` used in the expression, a `val` argument needs to be passed
|
||||
>>> response.xpath('//div[@id=$val]/a/text()', val='images').get()
|
||||
'Name: My image 1 '
|
||||
>>> # `$val` used in the expression, a `val` argument needs to be passed
|
||||
>>> response.xpath('//div[@id=$val]/a/text()', val='images').get()
|
||||
'Name: My image 1 '
|
||||
|
||||
Here's another example, to find the "id" attribute of a ``<div>`` tag containing
|
||||
five ``<a>`` children (here we pass the value ``5`` as an integer)::
|
||||
five ``<a>`` children (here we pass the value ``5`` as an integer):
|
||||
|
||||
>>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get()
|
||||
'images'
|
||||
>>> response.xpath('//div[count(a)=$cnt]/@id', cnt=5).get()
|
||||
'images'
|
||||
|
||||
All variable references must have a binding value when calling ``.xpath()``
|
||||
(otherwise you'll get a ``ValueError: XPath error:`` exception).
|
||||
|
|
@ -687,19 +685,19 @@ You can see several namespace declarations including a default
|
|||
.. highlight:: python
|
||||
|
||||
Once in the shell we can try selecting all ``<link>`` objects and see that it
|
||||
doesn't work (because the Atom XML namespace is obfuscating those nodes)::
|
||||
doesn't work (because the Atom XML namespace is obfuscating those nodes):
|
||||
|
||||
>>> response.xpath("//link")
|
||||
[]
|
||||
>>> response.xpath("//link")
|
||||
[]
|
||||
|
||||
But once we call the :meth:`Selector.remove_namespaces` method, all
|
||||
nodes can be accessed directly by their names::
|
||||
nodes can be accessed directly by their names:
|
||||
|
||||
>>> response.selector.remove_namespaces()
|
||||
>>> response.xpath("//link")
|
||||
[<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>,
|
||||
<Selector xpath='//link' data='<link rel="next" type="application/atom+'>,
|
||||
...
|
||||
>>> response.selector.remove_namespaces()
|
||||
>>> response.xpath("//link")
|
||||
[<Selector xpath='//link' data='<link rel="alternate" type="text/html" h'>,
|
||||
<Selector xpath='//link' data='<link rel="next" type="application/atom+'>,
|
||||
...
|
||||
|
||||
If you wonder why the namespace removal procedure isn't always called by default
|
||||
instead of having to call it manually, this is because of two reasons, which, in order
|
||||
|
|
@ -734,26 +732,25 @@ Regular expressions
|
|||
The ``test()`` function, for example, can prove quite useful when XPath's
|
||||
``starts-with()`` or ``contains()`` are not sufficient.
|
||||
|
||||
Example selecting links in list item with a "class" attribute ending with a digit::
|
||||
Example selecting links in list item with a "class" attribute ending with a digit:
|
||||
|
||||
>>> from scrapy import Selector
|
||||
>>> doc = u"""
|
||||
... <div>
|
||||
... <ul>
|
||||
... <li class="item-0"><a href="link1.html">first item</a></li>
|
||||
... <li class="item-1"><a href="link2.html">second item</a></li>
|
||||
... <li class="item-inactive"><a href="link3.html">third item</a></li>
|
||||
... <li class="item-1"><a href="link4.html">fourth item</a></li>
|
||||
... <li class="item-0"><a href="link5.html">fifth item</a></li>
|
||||
... </ul>
|
||||
... </div>
|
||||
... """
|
||||
>>> sel = Selector(text=doc, type="html")
|
||||
>>> sel.xpath('//li//@href').getall()
|
||||
['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html']
|
||||
>>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall()
|
||||
['link1.html', 'link2.html', 'link4.html', 'link5.html']
|
||||
>>>
|
||||
>>> from scrapy import Selector
|
||||
>>> doc = u"""
|
||||
... <div>
|
||||
... <ul>
|
||||
... <li class="item-0"><a href="link1.html">first item</a></li>
|
||||
... <li class="item-1"><a href="link2.html">second item</a></li>
|
||||
... <li class="item-inactive"><a href="link3.html">third item</a></li>
|
||||
... <li class="item-1"><a href="link4.html">fourth item</a></li>
|
||||
... <li class="item-0"><a href="link5.html">fifth item</a></li>
|
||||
... </ul>
|
||||
... </div>
|
||||
... """
|
||||
>>> sel = Selector(text=doc, type="html")
|
||||
>>> sel.xpath('//li//@href').getall()
|
||||
['link1.html', 'link2.html', 'link3.html', 'link4.html', 'link5.html']
|
||||
>>> sel.xpath('//li[re:test(@class, "item-\d$")]//@href').getall()
|
||||
['link1.html', 'link2.html', 'link4.html', 'link5.html']
|
||||
|
||||
.. warning:: C library ``libxslt`` doesn't natively support EXSLT regular
|
||||
expressions so `lxml`_'s implementation uses hooks to Python's ``re`` module.
|
||||
|
|
@ -766,7 +763,7 @@ Set operations
|
|||
These can be handy for excluding parts of a document tree before
|
||||
extracting text elements for example.
|
||||
|
||||
Example extracting microdata (sample content taken from http://schema.org/Product)
|
||||
Example extracting microdata (sample content taken from https://schema.org/Product)
|
||||
with groups of itemscopes and corresponding itemprops::
|
||||
|
||||
>>> doc = u"""
|
||||
|
|
@ -849,7 +846,6 @@ with groups of itemscopes and corresponding itemprops::
|
|||
current scope: ['http://schema.org/Rating']
|
||||
properties: ['worstRating', 'ratingValue', 'bestRating']
|
||||
|
||||
>>>
|
||||
|
||||
Here we first iterate over ``itemscope`` elements, and for each one,
|
||||
we look for all ``itemprops`` elements and exclude those that are themselves
|
||||
|
|
@ -877,15 +873,15 @@ For the following HTML::
|
|||
|
||||
.. highlight:: python
|
||||
|
||||
You can use it like this::
|
||||
You can use it like this:
|
||||
|
||||
>>> response.xpath('//p[has-class("foo")]')
|
||||
[<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>,
|
||||
<Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar-baz")]')
|
||||
[<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar")]')
|
||||
[]
|
||||
>>> response.xpath('//p[has-class("foo")]')
|
||||
[<Selector xpath='//p[has-class("foo")]' data='<p class="foo bar-baz">First</p>'>,
|
||||
<Selector xpath='//p[has-class("foo")]' data='<p class="foo">Second</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar-baz")]')
|
||||
[<Selector xpath='//p[has-class("foo", "bar-baz")]' data='<p class="foo bar-baz">First</p>'>]
|
||||
>>> response.xpath('//p[has-class("foo", "bar")]')
|
||||
[]
|
||||
|
||||
So XPath ``//p[has-class("foo", "bar-baz")]`` is roughly equivalent to CSS
|
||||
``p.foo.bar-baz``. Please note, that it is slower in most of the cases,
|
||||
|
|
@ -989,7 +985,7 @@ a :class:`~scrapy.http.HtmlResponse` object like this::
|
|||
sel = Selector(html_response)
|
||||
|
||||
1. Select all ``<h1>`` elements from an HTML response body, returning a list of
|
||||
:class:`Selector` objects (ie. a :class:`SelectorList` object)::
|
||||
:class:`Selector` objects (i.e. a :class:`SelectorList` object)::
|
||||
|
||||
sel.xpath("//h1")
|
||||
|
||||
|
|
@ -1016,7 +1012,7 @@ instantiated with an :class:`~scrapy.http.XmlResponse` object::
|
|||
sel = Selector(xml_response)
|
||||
|
||||
1. Select all ``<product>`` elements from an XML response body, returning a list
|
||||
of :class:`Selector` objects (ie. a :class:`SelectorList` object)::
|
||||
of :class:`Selector` objects (i.e. a :class:`SelectorList` object)::
|
||||
|
||||
sel.xpath("//product")
|
||||
|
||||
|
|
|
|||
|
|
@ -188,7 +188,6 @@ AWS_ENDPOINT_URL
|
|||
Default: ``None``
|
||||
|
||||
Endpoint URL used for S3-like storage, for example Minio or s3.scality.
|
||||
Only supported with ``botocore`` library.
|
||||
|
||||
.. setting:: AWS_USE_SSL
|
||||
|
||||
|
|
@ -199,7 +198,6 @@ Default: ``None``
|
|||
|
||||
Use this option if you want to disable SSL connection for communication with
|
||||
S3 or S3-like storage. By default SSL will be used.
|
||||
Only supported with ``botocore`` library.
|
||||
|
||||
.. setting:: AWS_VERIFY
|
||||
|
||||
|
|
@ -209,7 +207,7 @@ AWS_VERIFY
|
|||
Default: ``None``
|
||||
|
||||
Verify SSL connection between Scrapy and S3 or S3-like storage. By default
|
||||
SSL verification will occur. Only supported with ``botocore`` library.
|
||||
SSL verification will occur.
|
||||
|
||||
.. setting:: AWS_REGION_NAME
|
||||
|
||||
|
|
@ -219,7 +217,6 @@ AWS_REGION_NAME
|
|||
Default: ``None``
|
||||
|
||||
The name of the region associated with the AWS client.
|
||||
Only supported with ``botocore`` library.
|
||||
|
||||
.. setting:: BOT_NAME
|
||||
|
||||
|
|
@ -251,7 +248,7 @@ CONCURRENT_REQUESTS
|
|||
|
||||
Default: ``16``
|
||||
|
||||
The maximum number of concurrent (ie. simultaneous) requests that will be
|
||||
The maximum number of concurrent (i.e. simultaneous) requests that will be
|
||||
performed by the Scrapy downloader.
|
||||
|
||||
.. setting:: CONCURRENT_REQUESTS_PER_DOMAIN
|
||||
|
|
@ -261,7 +258,7 @@ CONCURRENT_REQUESTS_PER_DOMAIN
|
|||
|
||||
Default: ``8``
|
||||
|
||||
The maximum number of concurrent (ie. simultaneous) requests that will be
|
||||
The maximum number of concurrent (i.e. simultaneous) requests that will be
|
||||
performed to any single domain.
|
||||
|
||||
See also: :ref:`topics-autothrottle` and its
|
||||
|
|
@ -275,7 +272,7 @@ CONCURRENT_REQUESTS_PER_IP
|
|||
|
||||
Default: ``0``
|
||||
|
||||
The maximum number of concurrent (ie. simultaneous) requests that will be
|
||||
The maximum number of concurrent (i.e. simultaneous) requests that will be
|
||||
performed to any single IP. If non-zero, the
|
||||
:setting:`CONCURRENT_REQUESTS_PER_DOMAIN` setting is ignored, and this one is
|
||||
used instead. In other words, concurrency limits will be applied per IP, not
|
||||
|
|
@ -379,6 +376,19 @@ Default: ``10000``
|
|||
|
||||
DNS in-memory cache size.
|
||||
|
||||
.. setting:: DNS_RESOLVER
|
||||
|
||||
DNS_RESOLVER
|
||||
------------
|
||||
|
||||
Default: ``'scrapy.resolver.CachingThreadedResolver'``
|
||||
|
||||
The class to be used to resolve DNS names. The default ``scrapy.resolver.CachingThreadedResolver``
|
||||
supports specifying a timeout for DNS requests via the :setting:`DNS_TIMEOUT` setting,
|
||||
but works only with IPv4 addresses. Scrapy provides an alternative resolver,
|
||||
``scrapy.resolver.CachingHostnameResolver``, which supports IPv4/IPv6 addresses but does not
|
||||
take the :setting:`DNS_TIMEOUT` setting into account.
|
||||
|
||||
.. setting:: DNS_TIMEOUT
|
||||
|
||||
DNS_TIMEOUT
|
||||
|
|
@ -887,7 +897,7 @@ LOG_FORMAT
|
|||
|
||||
Default: ``'%(asctime)s [%(name)s] %(levelname)s: %(message)s'``
|
||||
|
||||
String for formatting log messsages. Refer to the `Python logging documentation`_ for the whole list of available
|
||||
String for formatting log messages. Refer to the `Python logging documentation`_ for the whole list of available
|
||||
placeholders.
|
||||
|
||||
.. _Python logging documentation: https://docs.python.org/2/library/logging.html#logrecord-attributes
|
||||
|
|
@ -1244,6 +1254,17 @@ Type of priority queue used by the scheduler. Another available type is
|
|||
domains in parallel. But currently ``scrapy.pqueues.DownloaderAwarePriorityQueue``
|
||||
does not work together with :setting:`CONCURRENT_REQUESTS_PER_IP`.
|
||||
|
||||
.. setting:: SCRAPER_SLOT_MAX_ACTIVE_SIZE
|
||||
|
||||
SCRAPER_SLOT_MAX_ACTIVE_SIZE
|
||||
----------------------------
|
||||
Default: ``5_000_000``
|
||||
|
||||
Soft limit (in bytes) for response data being processed.
|
||||
|
||||
While the sum of the sizes of all responses being processed is above this value,
|
||||
Scrapy does not process new requests.
|
||||
|
||||
.. setting:: SPIDER_CONTRACTS
|
||||
|
||||
SPIDER_CONTRACTS
|
||||
|
|
@ -1267,7 +1288,7 @@ Default::
|
|||
'scrapy.contracts.default.ScrapesContract': 3,
|
||||
}
|
||||
|
||||
A dict containing the scrapy contracts enabled by default in Scrapy. You should
|
||||
A dict containing the Scrapy contracts enabled by default in Scrapy. You should
|
||||
never modify this setting in your project, modify :setting:`SPIDER_CONTRACTS`
|
||||
instead. For more info see :ref:`topics-contracts`.
|
||||
|
||||
|
|
@ -1298,7 +1319,7 @@ SPIDER_LOADER_WARN_ONLY
|
|||
|
||||
Default: ``False``
|
||||
|
||||
By default, when scrapy tries to import spider classes from :setting:`SPIDER_MODULES`,
|
||||
By default, when Scrapy tries to import spider classes from :setting:`SPIDER_MODULES`,
|
||||
it will fail loudly if there is any ``ImportError`` exception.
|
||||
But you can choose to silence this exception and turn it into a simple
|
||||
warning by setting ``SPIDER_LOADER_WARN_ONLY = True``.
|
||||
|
|
@ -1421,6 +1442,30 @@ command.
|
|||
The project name must not conflict with the name of custom files or directories
|
||||
in the ``project`` subdirectory.
|
||||
|
||||
.. setting:: TWISTED_REACTOR
|
||||
|
||||
TWISTED_REACTOR
|
||||
---------------
|
||||
|
||||
Default: ``None``
|
||||
|
||||
Import path of a given Twisted reactor, for instance:
|
||||
:class:`twisted.internet.asyncioreactor.AsyncioSelectorReactor`.
|
||||
|
||||
Scrapy will install this reactor if no other is installed yet, such as when
|
||||
the ``scrapy`` CLI program is invoked or when using the
|
||||
:class:`~scrapy.crawler.CrawlerProcess` class. If you are using the
|
||||
:class:`~scrapy.crawler.CrawlerRunner` class, you need to install the correct
|
||||
reactor manually. An exception will be raised if the installation fails.
|
||||
|
||||
The default value for this option is currently ``None``, which means that Scrapy
|
||||
will not attempt to install any specific reactor, and the default one defined by
|
||||
Twisted for the current platform will be used. This is to maintain backward
|
||||
compatibility and avoid possible problems caused by using a non-default reactor.
|
||||
|
||||
For additional information, please see
|
||||
:doc:`core/howto/choosing-reactor`.
|
||||
|
||||
|
||||
.. setting:: URLLENGTH_LIMIT
|
||||
|
||||
|
|
|
|||
|
|
@ -31,7 +31,7 @@ for more info.
|
|||
Scrapy also has support for `bpython`_, and will try to use it where `IPython`_
|
||||
is unavailable.
|
||||
|
||||
Through scrapy's settings you can configure it to use any one of
|
||||
Through Scrapy's settings you can configure it to use any one of
|
||||
``ipython``, ``bpython`` or the standard ``python`` shell, regardless of which
|
||||
are installed. This is done by setting the ``SCRAPY_PYTHON_SHELL`` environment
|
||||
variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`::
|
||||
|
|
@ -41,7 +41,7 @@ variable; or by defining it in your :ref:`scrapy.cfg <topics-config-settings>`::
|
|||
|
||||
.. _IPython: https://ipython.org/
|
||||
.. _IPython installation guide: https://ipython.org/install.html
|
||||
.. _bpython: https://www.bpython-interpreter.org/
|
||||
.. _bpython: https://bpython-interpreter.org/
|
||||
|
||||
Launch the shell
|
||||
================
|
||||
|
|
@ -142,7 +142,7 @@ Example of shell session
|
|||
========================
|
||||
|
||||
Here's an example of a typical shell session where we start by scraping the
|
||||
https://scrapy.org page, and then proceed to scrape the https://reddit.com
|
||||
https://scrapy.org page, and then proceed to scrape the https://old.reddit.com/
|
||||
page. Finally, we modify the (Reddit) request method to POST and re-fetch it
|
||||
getting an error. We end the session by typing Ctrl-D (in Unix systems) or
|
||||
Ctrl-Z in Windows.
|
||||
|
|
@ -177,47 +177,46 @@ all start with the ``[s]`` prefix)::
|
|||
>>>
|
||||
|
||||
|
||||
After that, we can start playing with the objects::
|
||||
After that, we can start playing with the objects:
|
||||
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'Scrapy | A Fast and Powerful Scraping and Web Crawling Framework'
|
||||
|
||||
>>> fetch("https://reddit.com")
|
||||
>>> fetch("https://old.reddit.com/")
|
||||
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'reddit: the front page of the internet'
|
||||
>>> response.xpath('//title/text()').get()
|
||||
'reddit: the front page of the internet'
|
||||
|
||||
>>> request = request.replace(method="POST")
|
||||
>>> request = request.replace(method="POST")
|
||||
|
||||
>>> fetch(request)
|
||||
>>> fetch(request)
|
||||
|
||||
>>> response.status
|
||||
404
|
||||
>>> response.status
|
||||
404
|
||||
|
||||
>>> from pprint import pprint
|
||||
>>> from pprint import pprint
|
||||
|
||||
>>> pprint(response.headers)
|
||||
{'Accept-Ranges': ['bytes'],
|
||||
'Cache-Control': ['max-age=0, must-revalidate'],
|
||||
'Content-Type': ['text/html; charset=UTF-8'],
|
||||
'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'],
|
||||
'Server': ['snooserv'],
|
||||
'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'],
|
||||
'Vary': ['accept-encoding'],
|
||||
'Via': ['1.1 varnish'],
|
||||
'X-Cache': ['MISS'],
|
||||
'X-Cache-Hits': ['0'],
|
||||
'X-Content-Type-Options': ['nosniff'],
|
||||
'X-Frame-Options': ['SAMEORIGIN'],
|
||||
'X-Moose': ['majestic'],
|
||||
'X-Served-By': ['cache-cdg8730-CDG'],
|
||||
'X-Timer': ['S1481214079.394283,VS0,VE159'],
|
||||
'X-Ua-Compatible': ['IE=edge'],
|
||||
'X-Xss-Protection': ['1; mode=block']}
|
||||
>>>
|
||||
>>> pprint(response.headers)
|
||||
{'Accept-Ranges': ['bytes'],
|
||||
'Cache-Control': ['max-age=0, must-revalidate'],
|
||||
'Content-Type': ['text/html; charset=UTF-8'],
|
||||
'Date': ['Thu, 08 Dec 2016 16:21:19 GMT'],
|
||||
'Server': ['snooserv'],
|
||||
'Set-Cookie': ['loid=KqNLou0V9SKMX4qb4n; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.445Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loid=vi0ZVe4NkxNWdlH7r7; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure',
|
||||
'loidcreated=2016-12-08T16%3A21%3A19.459Z; Domain=reddit.com; Max-Age=63071999; Path=/; expires=Sat, 08-Dec-2018 16:21:19 GMT; secure'],
|
||||
'Vary': ['accept-encoding'],
|
||||
'Via': ['1.1 varnish'],
|
||||
'X-Cache': ['MISS'],
|
||||
'X-Cache-Hits': ['0'],
|
||||
'X-Content-Type-Options': ['nosniff'],
|
||||
'X-Frame-Options': ['SAMEORIGIN'],
|
||||
'X-Moose': ['majestic'],
|
||||
'X-Served-By': ['cache-cdg8730-CDG'],
|
||||
'X-Timer': ['S1481214079.394283,VS0,VE159'],
|
||||
'X-Ua-Compatible': ['IE=edge'],
|
||||
'X-Xss-Protection': ['1; mode=block']}
|
||||
|
||||
|
||||
.. _topics-shell-inspect-response:
|
||||
|
|
@ -263,16 +262,16 @@ When you run the spider, you will get something similar to this::
|
|||
>>> response.url
|
||||
'http://example.org'
|
||||
|
||||
Then, you can check if the extraction code is working::
|
||||
Then, you can check if the extraction code is working:
|
||||
|
||||
>>> response.xpath('//h1[@class="fn"]')
|
||||
[]
|
||||
>>> response.xpath('//h1[@class="fn"]')
|
||||
[]
|
||||
|
||||
Nope, it doesn't. So you can open the response in your web browser and see if
|
||||
it's the response you were expecting::
|
||||
it's the response you were expecting:
|
||||
|
||||
>>> view(response)
|
||||
True
|
||||
>>> view(response)
|
||||
True
|
||||
|
||||
Finally you hit Ctrl-D (or Ctrl-Z in Windows) to exit the shell and resume the
|
||||
crawling::
|
||||
|
|
|
|||
|
|
@ -50,10 +50,10 @@ Here is a simple example showing how you can catch signals and perform some acti
|
|||
Deferred signal handlers
|
||||
========================
|
||||
|
||||
Some signals support returning `Twisted deferreds`_ from their handlers, see
|
||||
the :ref:`topics-signals-ref` below to know which ones.
|
||||
Some signals support returning :class:`~twisted.internet.defer.Deferred`
|
||||
objects from their handlers, see the :ref:`topics-signals-ref` below to know
|
||||
which ones.
|
||||
|
||||
.. _Twisted deferreds: https://twistedmatrix.com/documents/current/core/howto/defer.html
|
||||
|
||||
.. _topics-signals-ref:
|
||||
|
||||
|
|
@ -141,7 +141,7 @@ item_error
|
|||
.. signal:: item_error
|
||||
.. function:: item_error(item, response, spider, failure)
|
||||
|
||||
Sent when a :ref:`topics-item-pipeline` generates an error (ie. raises
|
||||
Sent when a :ref:`topics-item-pipeline` generates an error (i.e. raises
|
||||
an exception), except :exc:`~scrapy.exceptions.DropItem` exception.
|
||||
|
||||
This signal supports returning deferreds from their handlers.
|
||||
|
|
@ -155,8 +155,8 @@ item_error
|
|||
:param spider: the spider which raised the exception
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
:param failure: the exception raised as a Twisted `Failure`_ object
|
||||
:type failure: `Failure`_ object
|
||||
:param failure: the exception raised
|
||||
:type failure: twisted.python.failure.Failure
|
||||
|
||||
spider_closed
|
||||
-------------
|
||||
|
|
@ -232,12 +232,12 @@ spider_error
|
|||
.. signal:: spider_error
|
||||
.. function:: spider_error(failure, response, spider)
|
||||
|
||||
Sent when a spider callback generates an error (ie. raises an exception).
|
||||
Sent when a spider callback generates an error (i.e. raises an exception).
|
||||
|
||||
This signal does not support returning deferreds from their handlers.
|
||||
|
||||
:param failure: the exception raised as a Twisted `Failure`_ object
|
||||
:type failure: `Failure`_ object
|
||||
:param failure: the exception raised
|
||||
:type failure: twisted.python.failure.Failure
|
||||
|
||||
:param response: the response being processed when the exception was raised
|
||||
:type response: :class:`~scrapy.http.Response` object
|
||||
|
|
@ -295,6 +295,23 @@ request_reached_downloader
|
|||
:param spider: the spider that yielded the request
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
request_left_downloader
|
||||
-----------------------
|
||||
|
||||
.. signal:: request_left_downloader
|
||||
.. function:: request_left_downloader(request, spider)
|
||||
|
||||
Sent when a :class:`~scrapy.http.Request` leaves the downloader, even in case of
|
||||
failure.
|
||||
|
||||
This signal does not support returning deferreds from its handlers.
|
||||
|
||||
:param request: the request that reached the downloader
|
||||
:type request: :class:`~scrapy.http.Request` object
|
||||
|
||||
:param spider: the spider that yielded the request
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
response_received
|
||||
-----------------
|
||||
|
||||
|
|
@ -333,5 +350,3 @@ response_downloaded
|
|||
|
||||
:param spider: the spider for which the response is intended
|
||||
:type spider: :class:`~scrapy.spiders.Spider` object
|
||||
|
||||
.. _Failure: https://twistedmatrix.com/documents/current/api/twisted.python.failure.Failure.html
|
||||
|
|
|
|||
|
|
@ -72,8 +72,6 @@ scrapy.Spider
|
|||
spider that crawls ``mywebsite.com`` would often be called
|
||||
``mywebsite``.
|
||||
|
||||
.. note:: In Python 2 this must be ASCII only.
|
||||
|
||||
.. attribute:: allowed_domains
|
||||
|
||||
An optional list of strings containing domains that this spider is
|
||||
|
|
@ -301,8 +299,8 @@ The spider will not do any parsing on its own.
|
|||
If you were to set the ``start_urls`` attribute from the command line,
|
||||
you would have to parse it on your own into a list
|
||||
using something like
|
||||
`ast.literal_eval <https://docs.python.org/library/ast.html#ast.literal_eval>`_
|
||||
or `json.loads <https://docs.python.org/library/json.html#json.loads>`_
|
||||
`ast.literal_eval <https://docs.python.org/3/library/ast.html#ast.literal_eval>`_
|
||||
or `json.loads <https://docs.python.org/3/library/json.html#json.loads>`_
|
||||
and then set it as an attribute.
|
||||
Otherwise, you would cause iteration over a ``start_urls`` string
|
||||
(a very common python pitfall)
|
||||
|
|
@ -416,6 +414,12 @@ Crawling rules
|
|||
from which the request originated as second argument. It must return a
|
||||
``Request`` object or ``None`` (to filter out the request).
|
||||
|
||||
``errback`` is a callable or a string (in which case a method from the spider
|
||||
object with that name will be used) to be called if any exception is
|
||||
raised while processing a request generated by the rule.
|
||||
It receives a :class:`Twisted Failure <twisted.python.failure.Failure>`
|
||||
instance as first parameter.
|
||||
|
||||
CrawlSpider example
|
||||
~~~~~~~~~~~~~~~~~~~
|
||||
|
||||
|
|
@ -807,6 +811,6 @@ Combine SitemapSpider with other sources of urls::
|
|||
|
||||
.. _Sitemaps: https://www.sitemaps.org/index.html
|
||||
.. _Sitemap index files: https://www.sitemaps.org/protocol.html#index
|
||||
.. _robots.txt: http://www.robotstxt.org/
|
||||
.. _robots.txt: https://www.robotstxt.org/
|
||||
.. _TLD: https://en.wikipedia.org/wiki/Top-level_domain
|
||||
.. _Scrapyd documentation: https://scrapyd.readthedocs.io/en/latest/
|
||||
|
|
|
|||
|
|
@ -57,15 +57,15 @@ Set stat value only if lower than previous::
|
|||
|
||||
stats.min_value('min_free_memory_percent', value)
|
||||
|
||||
Get stat value::
|
||||
Get stat value:
|
||||
|
||||
>>> stats.get_value('custom_count')
|
||||
1
|
||||
>>> stats.get_value('custom_count')
|
||||
1
|
||||
|
||||
Get all stats::
|
||||
Get all stats:
|
||||
|
||||
>>> stats.get_stats()
|
||||
{'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)}
|
||||
>>> stats.get_stats()
|
||||
{'custom_count': 1, 'start_time': datetime.datetime(2009, 7, 14, 21, 47, 28, 977139)}
|
||||
|
||||
Available Stats Collectors
|
||||
==========================
|
||||
|
|
|
|||
|
|
@ -44,11 +44,11 @@ the console you need to type::
|
|||
>>>
|
||||
|
||||
By default Username is ``scrapy`` and Password is autogenerated. The
|
||||
autogenerated Password can be seen on scrapy logs like the example below::
|
||||
autogenerated Password can be seen on Scrapy logs like the example below::
|
||||
|
||||
2018-10-16 14:35:21 [scrapy.extensions.telnet] INFO: Telnet Password: 16f92501e8a59326
|
||||
|
||||
Default Username and Password can be overriden by the settings
|
||||
Default Username and Password can be overridden by the settings
|
||||
:setting:`TELNETCONSOLE_USERNAME` and :setting:`TELNETCONSOLE_PASSWORD`.
|
||||
|
||||
.. warning::
|
||||
|
|
|
|||
|
|
@ -1,5 +1,4 @@
|
|||
#!/usr/bin/env python
|
||||
from __future__ import print_function
|
||||
from time import time
|
||||
from collections import deque
|
||||
from twisted.web.server import Site, NOT_DONE_YET
|
||||
|
|
|
|||
|
|
@ -1,11 +1,12 @@
|
|||
#compdef scrapy
|
||||
_scrapy() {
|
||||
local context state state_descr line
|
||||
local ret=1
|
||||
typeset -A opt_args
|
||||
_arguments \
|
||||
"(- 1 *)--help[Help]" \
|
||||
"(- 1 *)"{-h,--help}"[Help]" \
|
||||
"1: :->command" \
|
||||
"*:: :->args"
|
||||
"*:: :->args" && ret=0
|
||||
|
||||
case $state in
|
||||
command)
|
||||
|
|
@ -61,7 +62,7 @@ _scrapy() {
|
|||
'-c[evaluate the code in the shell, print the result and exit]:code:(CODE)'
|
||||
'--no-redirect[do not handle HTTP 3xx status codes and print response as-is]'
|
||||
'--spider[use this spider]:spider:_scrapy_spiders'
|
||||
'::file:_files -g \*.http'
|
||||
'::file:_files -g \*.html'
|
||||
'::URL:_httpie_urls'
|
||||
)
|
||||
_scrapy_glb_opts $options
|
||||
|
|
@ -134,6 +135,8 @@ _scrapy() {
|
|||
esac
|
||||
;;
|
||||
esac
|
||||
|
||||
return ret
|
||||
}
|
||||
|
||||
_scrapy_cmds() {
|
||||
|
|
|
|||
247
pytest.ini
247
pytest.ini
|
|
@ -2,5 +2,250 @@
|
|||
usefixtures = chdir
|
||||
python_files=test_*.py __init__.py
|
||||
python_classes=
|
||||
addopts = --doctest-modules --assert=plain
|
||||
addopts =
|
||||
--assert=plain
|
||||
--doctest-modules
|
||||
--ignore=docs/_ext
|
||||
--ignore=docs/conf.py
|
||||
--ignore=docs/news.rst
|
||||
--ignore=docs/topics/dynamic-content.rst
|
||||
--ignore=docs/topics/items.rst
|
||||
--ignore=docs/topics/leaks.rst
|
||||
--ignore=docs/topics/loaders.rst
|
||||
--ignore=docs/topics/selectors.rst
|
||||
--ignore=docs/topics/shell.rst
|
||||
--ignore=docs/topics/stats.rst
|
||||
--ignore=docs/topics/telnetconsole.rst
|
||||
--ignore=docs/utils
|
||||
twisted = 1
|
||||
markers =
|
||||
only_asyncio: marks tests as only enabled when --reactor=asyncio is passed
|
||||
flake8-ignore =
|
||||
W503
|
||||
# Files that are only meant to provide top-level imports are expected not
|
||||
# to use any of their imports:
|
||||
scrapy/core/downloader/handlers/http.py F401
|
||||
scrapy/http/__init__.py F401
|
||||
# Issues pending a review:
|
||||
# extras
|
||||
extras/qps-bench-server.py E501
|
||||
extras/qpsclient.py E501 E501
|
||||
# scrapy/commands
|
||||
scrapy/commands/__init__.py E128 E501
|
||||
scrapy/commands/check.py E501
|
||||
scrapy/commands/crawl.py E501
|
||||
scrapy/commands/edit.py E501
|
||||
scrapy/commands/fetch.py E401 E501 E128 E731
|
||||
scrapy/commands/genspider.py E128 E501 E502
|
||||
scrapy/commands/parse.py E128 E501 E731
|
||||
scrapy/commands/runspider.py E501
|
||||
scrapy/commands/settings.py E128
|
||||
scrapy/commands/shell.py E128 E501 E502
|
||||
scrapy/commands/startproject.py E127 E501 E128
|
||||
scrapy/commands/version.py E501 E128
|
||||
# scrapy/contracts
|
||||
scrapy/contracts/__init__.py E501 W504
|
||||
scrapy/contracts/default.py E128
|
||||
# scrapy/core
|
||||
scrapy/core/engine.py E501 E128 E127 E502
|
||||
scrapy/core/scheduler.py E501
|
||||
scrapy/core/scraper.py E501 E128 W504
|
||||
scrapy/core/spidermw.py E501 E731 E126
|
||||
scrapy/core/downloader/__init__.py E501
|
||||
scrapy/core/downloader/contextfactory.py E501 E128 E126
|
||||
scrapy/core/downloader/middleware.py E501 E502
|
||||
scrapy/core/downloader/tls.py E501 E241
|
||||
scrapy/core/downloader/webclient.py E731 E501 E128 E126
|
||||
scrapy/core/downloader/handlers/__init__.py E501
|
||||
scrapy/core/downloader/handlers/ftp.py E501 E128 E127
|
||||
scrapy/core/downloader/handlers/http10.py E501
|
||||
scrapy/core/downloader/handlers/http11.py E501
|
||||
scrapy/core/downloader/handlers/s3.py E501 E128 E126
|
||||
# scrapy/downloadermiddlewares
|
||||
scrapy/downloadermiddlewares/ajaxcrawl.py E501
|
||||
scrapy/downloadermiddlewares/decompression.py E501
|
||||
scrapy/downloadermiddlewares/defaultheaders.py E501
|
||||
scrapy/downloadermiddlewares/httpcache.py E501 E126
|
||||
scrapy/downloadermiddlewares/httpcompression.py E501 E128
|
||||
scrapy/downloadermiddlewares/httpproxy.py E501
|
||||
scrapy/downloadermiddlewares/redirect.py E501 W504
|
||||
scrapy/downloadermiddlewares/retry.py E501 E126
|
||||
scrapy/downloadermiddlewares/robotstxt.py E501
|
||||
scrapy/downloadermiddlewares/stats.py E501
|
||||
# scrapy/extensions
|
||||
scrapy/extensions/closespider.py E501 E128 E123
|
||||
scrapy/extensions/corestats.py E501
|
||||
scrapy/extensions/feedexport.py E128 E501
|
||||
scrapy/extensions/httpcache.py E128 E501
|
||||
scrapy/extensions/memdebug.py E501
|
||||
scrapy/extensions/spiderstate.py E501
|
||||
scrapy/extensions/telnet.py E501 W504
|
||||
scrapy/extensions/throttle.py E501
|
||||
# scrapy/http
|
||||
scrapy/http/common.py E501
|
||||
scrapy/http/cookies.py E501
|
||||
scrapy/http/request/__init__.py E501
|
||||
scrapy/http/request/form.py E501 E123
|
||||
scrapy/http/request/json_request.py E501
|
||||
scrapy/http/response/__init__.py E501 E128
|
||||
scrapy/http/response/text.py E501 E128 E124
|
||||
# scrapy/linkextractors
|
||||
scrapy/linkextractors/__init__.py E731 E501 E402 W504
|
||||
scrapy/linkextractors/lxmlhtml.py E501 E731
|
||||
# scrapy/loader
|
||||
scrapy/loader/__init__.py E501 E128
|
||||
scrapy/loader/processors.py E501
|
||||
# scrapy/pipelines
|
||||
scrapy/pipelines/__init__.py E501
|
||||
scrapy/pipelines/files.py E116 E501 E266
|
||||
scrapy/pipelines/images.py E265 E501
|
||||
scrapy/pipelines/media.py E125 E501 E266
|
||||
# scrapy/selector
|
||||
scrapy/selector/__init__.py F403
|
||||
scrapy/selector/unified.py E501 E111
|
||||
# scrapy/settings
|
||||
scrapy/settings/__init__.py E501
|
||||
scrapy/settings/default_settings.py E501 E114 E116
|
||||
scrapy/settings/deprecated.py E501
|
||||
# scrapy/spidermiddlewares
|
||||
scrapy/spidermiddlewares/httperror.py E501
|
||||
scrapy/spidermiddlewares/offsite.py E501
|
||||
scrapy/spidermiddlewares/referer.py E501 E129 W504
|
||||
scrapy/spidermiddlewares/urllength.py E501
|
||||
# scrapy/spiders
|
||||
scrapy/spiders/__init__.py E501 E402
|
||||
scrapy/spiders/crawl.py E501
|
||||
scrapy/spiders/feed.py E501
|
||||
scrapy/spiders/sitemap.py E501
|
||||
# scrapy/utils
|
||||
scrapy/utils/asyncio.py E501
|
||||
scrapy/utils/benchserver.py E501
|
||||
scrapy/utils/conf.py E402 E501
|
||||
scrapy/utils/datatypes.py E501
|
||||
scrapy/utils/decorators.py E501
|
||||
scrapy/utils/defer.py E501 E128
|
||||
scrapy/utils/deprecate.py E128 E501 E127 E502
|
||||
scrapy/utils/gz.py E501 W504
|
||||
scrapy/utils/http.py F403
|
||||
scrapy/utils/httpobj.py E501
|
||||
scrapy/utils/iterators.py E501
|
||||
scrapy/utils/log.py E128 E501
|
||||
scrapy/utils/markup.py F403
|
||||
scrapy/utils/misc.py E501
|
||||
scrapy/utils/multipart.py F403
|
||||
scrapy/utils/project.py E501
|
||||
scrapy/utils/python.py E501
|
||||
scrapy/utils/reactor.py E501
|
||||
scrapy/utils/reqser.py E501
|
||||
scrapy/utils/request.py E127 E501
|
||||
scrapy/utils/response.py E501 E128
|
||||
scrapy/utils/signal.py E501 E128
|
||||
scrapy/utils/sitemap.py E501
|
||||
scrapy/utils/spider.py E501
|
||||
scrapy/utils/ssl.py E501
|
||||
scrapy/utils/test.py E501
|
||||
scrapy/utils/url.py E501 F403 E128 F405
|
||||
# scrapy
|
||||
scrapy/__init__.py E402 E501
|
||||
scrapy/cmdline.py E501
|
||||
scrapy/crawler.py E501
|
||||
scrapy/dupefilters.py E501 E202
|
||||
scrapy/exceptions.py E501
|
||||
scrapy/exporters.py E501
|
||||
scrapy/interfaces.py E501
|
||||
scrapy/item.py E501 E128
|
||||
scrapy/link.py E501
|
||||
scrapy/logformatter.py E501
|
||||
scrapy/mail.py E402 E128 E501 E502
|
||||
scrapy/middleware.py E128 E501
|
||||
scrapy/pqueues.py E501
|
||||
scrapy/resolver.py E501
|
||||
scrapy/responsetypes.py E128 E501
|
||||
scrapy/robotstxt.py E501
|
||||
scrapy/shell.py E501
|
||||
scrapy/signalmanager.py E501
|
||||
scrapy/spiderloader.py F841 E501 E126
|
||||
scrapy/squeues.py E128
|
||||
scrapy/statscollectors.py E501
|
||||
# tests
|
||||
tests/__init__.py E402 E501
|
||||
tests/mockserver.py E401 E501 E126 E123
|
||||
tests/pipelines.py F841
|
||||
tests/spiders.py E501 E127
|
||||
tests/test_closespider.py E501 E127
|
||||
tests/test_command_fetch.py E501
|
||||
tests/test_command_parse.py E501 E128
|
||||
tests/test_command_shell.py E501 E128
|
||||
tests/test_commands.py E128 E501
|
||||
tests/test_contracts.py E501 E128
|
||||
tests/test_crawl.py E501 E741 E265
|
||||
tests/test_crawler.py F841 E501
|
||||
tests/test_dependencies.py F841 E501
|
||||
tests/test_downloader_handlers.py E124 E127 E128 E265 E501 E126 E123
|
||||
tests/test_downloadermiddleware.py E501
|
||||
tests/test_downloadermiddleware_ajaxcrawlable.py E501
|
||||
tests/test_downloadermiddleware_cookies.py E731 E741 E501 E128 E265 E126
|
||||
tests/test_downloadermiddleware_decompression.py E127
|
||||
tests/test_downloadermiddleware_defaultheaders.py E501
|
||||
tests/test_downloadermiddleware_downloadtimeout.py E501
|
||||
tests/test_downloadermiddleware_httpcache.py E501
|
||||
tests/test_downloadermiddleware_httpcompression.py E501 E126 E123
|
||||
tests/test_downloadermiddleware_httpproxy.py E501 E128
|
||||
tests/test_downloadermiddleware_redirect.py E501 E128 E127
|
||||
tests/test_downloadermiddleware_retry.py E501 E128 E126
|
||||
tests/test_downloadermiddleware_robotstxt.py E501
|
||||
tests/test_downloadermiddleware_stats.py E501
|
||||
tests/test_dupefilters.py E501 E741 E128 E124
|
||||
tests/test_engine.py E401 E501 E128
|
||||
tests/test_exporters.py E501 E731 E128 E124
|
||||
tests/test_extension_telnet.py F841
|
||||
tests/test_feedexport.py E501 F841 E241
|
||||
tests/test_http_cookies.py E501
|
||||
tests/test_http_headers.py E501
|
||||
tests/test_http_request.py E402 E501 E127 E128 E128 E126 E123
|
||||
tests/test_http_response.py E501 E128 E265
|
||||
tests/test_item.py E128 F841
|
||||
tests/test_link.py E501
|
||||
tests/test_linkextractors.py E501 E128 E124
|
||||
tests/test_loader.py E501 E731 E741 E128 E117 E241
|
||||
tests/test_logformatter.py E128 E501 E122
|
||||
tests/test_mail.py E128 E501
|
||||
tests/test_middleware.py E501 E128
|
||||
tests/test_pipeline_crawl.py E501 E128 E126
|
||||
tests/test_pipeline_files.py E501
|
||||
tests/test_pipeline_images.py F841 E501
|
||||
tests/test_pipeline_media.py E501 E741 E731 E128 E502
|
||||
tests/test_proxy_connect.py E501 E741
|
||||
tests/test_request_cb_kwargs.py E501
|
||||
tests/test_responsetypes.py E501
|
||||
tests/test_robotstxt_interface.py E501 E501
|
||||
tests/test_scheduler.py E501 E126 E123
|
||||
tests/test_selector.py E501 E127
|
||||
tests/test_spider.py E501
|
||||
tests/test_spidermiddleware.py E501
|
||||
tests/test_spidermiddleware_httperror.py E128 E501 E127 E121
|
||||
tests/test_spidermiddleware_offsite.py E501 E128 E111
|
||||
tests/test_spidermiddleware_output_chain.py E501
|
||||
tests/test_spidermiddleware_referer.py E501 F841 E125 E201 E124 E501 E241 E121
|
||||
tests/test_squeues.py E501 E741
|
||||
tests/test_utils_asyncio.py E501
|
||||
tests/test_utils_conf.py E501 E128
|
||||
tests/test_utils_curl.py E501
|
||||
tests/test_utils_datatypes.py E402 E501
|
||||
tests/test_utils_defer.py E501 F841
|
||||
tests/test_utils_deprecate.py F841 E501
|
||||
tests/test_utils_http.py E501 E128 W504
|
||||
tests/test_utils_iterators.py E501 E128 E129 E241
|
||||
tests/test_utils_log.py E741
|
||||
tests/test_utils_python.py E501 E731
|
||||
tests/test_utils_reqser.py E501 E128
|
||||
tests/test_utils_request.py E501 E128
|
||||
tests/test_utils_response.py E501
|
||||
tests/test_utils_signal.py E741 F841 E731
|
||||
tests/test_utils_sitemap.py E128 E501 E124
|
||||
tests/test_utils_url.py E501 E127 E125 E501 E241 E126 E123
|
||||
tests/test_webclient.py E501 E128 E122 E402 E241 E123 E126
|
||||
tests/test_cmdline/__init__.py E501
|
||||
tests/test_settings/__init__.py E501 E128
|
||||
tests/test_spiderloader/__init__.py E128 E501
|
||||
tests/test_utils_misc/__init__.py E501
|
||||
|
|
|
|||
|
|
@ -1,18 +0,0 @@
|
|||
parsel>=1.5.0
|
||||
PyDispatcher>=2.0.5
|
||||
w3lib>=1.17.0
|
||||
protego>=0.1.15
|
||||
|
||||
pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'"
|
||||
queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'"
|
||||
cryptography>=2.0 # Earlier versions would fail to install
|
||||
|
||||
# Reference versions taken from
|
||||
# https://packages.ubuntu.com/xenial/python/
|
||||
# https://packages.ubuntu.com/xenial/zope/
|
||||
cssselect>=0.9.1
|
||||
lxml>=3.5.0
|
||||
service_identity>=16.0.0
|
||||
six>=1.10.0
|
||||
Twisted>=16.0.0
|
||||
zope.interface>=4.1.3
|
||||
|
|
@ -1,18 +0,0 @@
|
|||
parsel>=1.5.0
|
||||
PyDispatcher>=2.0.5
|
||||
Twisted>=17.9.0
|
||||
w3lib>=1.17.0
|
||||
protego>=0.1.15
|
||||
|
||||
pyOpenSSL>=16.2.0 # Earlier versions fail with "AttributeError: module 'lib' has no attribute 'SSL_ST_INIT'"
|
||||
queuelib>=1.4.2 # Earlier versions fail with "AttributeError: '...QueueTest' object has no attribute 'qpath'"
|
||||
cryptography>=2.0 # Earlier versions would fail to install
|
||||
|
||||
# Reference versions taken from
|
||||
# https://packages.ubuntu.com/xenial/python/
|
||||
# https://packages.ubuntu.com/xenial/zope/
|
||||
cssselect>=0.9.1
|
||||
lxml>=3.5.0
|
||||
service_identity>=16.0.0
|
||||
six>=1.10.0
|
||||
zope.interface>=4.1.3
|
||||
|
|
@ -14,8 +14,8 @@ del pkgutil
|
|||
|
||||
# Check minimum required Python version
|
||||
import sys
|
||||
if sys.version_info < (2, 7):
|
||||
print("Scrapy %s requires Python 2.7" % __version__)
|
||||
if sys.version_info < (3, 5):
|
||||
print("Scrapy %s requires Python 3.5" % __version__)
|
||||
sys.exit(1)
|
||||
|
||||
# Ignore noisy twisted deprecation warnings
|
||||
|
|
@ -24,7 +24,7 @@ warnings.filterwarnings('ignore', category=DeprecationWarning, module='twisted')
|
|||
del warnings
|
||||
|
||||
# Apply monkey patches to fix issues in external libraries
|
||||
from . import _monkeypatches
|
||||
from scrapy import _monkeypatches
|
||||
del _monkeypatches
|
||||
|
||||
from twisted import version as _txv
|
||||
|
|
|
|||
|
|
@ -1,14 +1,4 @@
|
|||
import six
|
||||
from six.moves import copyreg
|
||||
|
||||
|
||||
if six.PY2:
|
||||
from urlparse import urlparse
|
||||
|
||||
# workaround for https://bugs.python.org/issue9374 - Python < 2.7.4
|
||||
if urlparse('s3://bucket/key?key=value').query != 'key=value':
|
||||
from urlparse import uses_query
|
||||
uses_query.append('s3')
|
||||
import copyreg
|
||||
|
||||
|
||||
# Undo what Twisted's perspective broker adds to pickle register
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
from __future__ import print_function
|
||||
import sys
|
||||
import os
|
||||
import optparse
|
||||
|
|
@ -68,7 +67,7 @@ def _pop_command_name(argv):
|
|||
|
||||
def _print_header(settings, inproject):
|
||||
if inproject:
|
||||
print("Scrapy %s - project: %s\n" % (scrapy.__version__, \
|
||||
print("Scrapy %s - project: %s\n" % (scrapy.__version__,
|
||||
settings['BOT_NAME']))
|
||||
else:
|
||||
print("Scrapy %s - no active project\n" % scrapy.__version__)
|
||||
|
|
@ -124,7 +123,7 @@ def execute(argv=None, settings=None):
|
|||
inproject = inside_project()
|
||||
cmds = _get_commands_dict(settings, inproject)
|
||||
cmdname = _pop_command_name(argv)
|
||||
parser = optparse.OptionParser(formatter=optparse.TitledHelpFormatter(), \
|
||||
parser = optparse.OptionParser(formatter=optparse.TitledHelpFormatter(),
|
||||
conflict_handler='resolve')
|
||||
if not cmdname:
|
||||
_print_commands(settings, inproject)
|
||||
|
|
|
|||
|
|
@ -1,8 +1,7 @@
|
|||
import sys
|
||||
import time
|
||||
import subprocess
|
||||
|
||||
from six.moves.urllib.parse import urlencode
|
||||
from urllib.parse import urlencode
|
||||
|
||||
import scrapy
|
||||
from scrapy.commands import ScrapyCommand
|
||||
|
|
|
|||
|
|
@ -1,6 +1,4 @@
|
|||
from __future__ import print_function
|
||||
import time
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from unittest import TextTestRunner, TextTestResult as _TextTestResult
|
||||
|
||||
|
|
@ -96,4 +94,3 @@ class Command(ScrapyCommand):
|
|||
result.printErrors()
|
||||
result.printSummary(start, stop)
|
||||
self.exitcode = int(not result.wasSuccessful())
|
||||
|
||||
|
|
|
|||
|
|
@ -54,8 +54,13 @@ class Command(ScrapyCommand):
|
|||
raise UsageError("running 'scrapy crawl' with more than one spider is no longer supported")
|
||||
spname = args[0]
|
||||
|
||||
self.crawler_process.crawl(spname, **opts.spargs)
|
||||
self.crawler_process.start()
|
||||
crawl_defer = self.crawler_process.crawl(spname, **opts.spargs)
|
||||
|
||||
if self.crawler_process.bootstrap_failed:
|
||||
if getattr(crawl_defer, 'result', None) is not None and issubclass(crawl_defer.result.type, Exception):
|
||||
self.exitcode = 1
|
||||
else:
|
||||
self.crawler_process.start()
|
||||
|
||||
if self.crawler_process.bootstrap_failed or \
|
||||
(hasattr(self.crawler_process, 'has_exception') and self.crawler_process.has_exception):
|
||||
self.exitcode = 1
|
||||
|
|
|
|||
|
|
@ -1,5 +1,4 @@
|
|||
from __future__ import print_function
|
||||
import sys, six
|
||||
import sys
|
||||
from w3lib.url import is_url
|
||||
|
||||
from scrapy.commands import ScrapyCommand
|
||||
|
|
@ -8,6 +7,7 @@ from scrapy.exceptions import UsageError
|
|||
from scrapy.utils.datatypes import SequenceExclude
|
||||
from scrapy.utils.spider import spidercls_for_request, DefaultSpider
|
||||
|
||||
|
||||
class Command(ScrapyCommand):
|
||||
|
||||
requires_project = False
|
||||
|
|
@ -24,12 +24,11 @@ class Command(ScrapyCommand):
|
|||
|
||||
def add_options(self, parser):
|
||||
ScrapyCommand.add_options(self, parser)
|
||||
parser.add_option("--spider", dest="spider",
|
||||
help="use this spider")
|
||||
parser.add_option("--headers", dest="headers", action="store_true", \
|
||||
help="print response HTTP headers instead of body")
|
||||
parser.add_option("--no-redirect", dest="no_redirect", action="store_true", \
|
||||
default=False, help="do not handle HTTP 3xx status codes and print response as-is")
|
||||
parser.add_option("--spider", dest="spider", help="use this spider")
|
||||
parser.add_option("--headers", dest="headers", action="store_true",
|
||||
help="print response HTTP headers instead of body")
|
||||
parser.add_option("--no-redirect", dest="no_redirect", action="store_true",
|
||||
default=False, help="do not handle HTTP 3xx status codes and print response as-is")
|
||||
|
||||
def _print_headers(self, headers, prefix):
|
||||
for key, values in headers.items():
|
||||
|
|
@ -45,8 +44,7 @@ class Command(ScrapyCommand):
|
|||
self._print_bytes(response.body)
|
||||
|
||||
def _print_bytes(self, bytes_):
|
||||
bytes_writer = sys.stdout if six.PY2 else sys.stdout.buffer
|
||||
bytes_writer.write(bytes_ + b'\n')
|
||||
sys.stdout.buffer.write(bytes_ + b'\n')
|
||||
|
||||
def run(self, args, opts):
|
||||
if len(args) != 1 or not is_url(args[0]):
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
from __future__ import print_function
|
||||
import os
|
||||
import shutil
|
||||
import string
|
||||
|
|
|
|||
|
|
@ -1,6 +1,6 @@
|
|||
from __future__ import print_function
|
||||
from scrapy.commands import ScrapyCommand
|
||||
|
||||
|
||||
class Command(ScrapyCommand):
|
||||
|
||||
requires_project = True
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
from __future__ import print_function
|
||||
import json
|
||||
import logging
|
||||
|
||||
|
|
@ -81,7 +80,7 @@ class Command(ScrapyCommand):
|
|||
else:
|
||||
items = self.items.get(lvl, [])
|
||||
|
||||
print("# Scraped Items ", "-"*60)
|
||||
print("# Scraped Items ", "-" * 60)
|
||||
display.pprint([dict(x) for x in items], colorize=colour)
|
||||
|
||||
def print_requests(self, lvl=None, colour=True):
|
||||
|
|
@ -93,14 +92,14 @@ class Command(ScrapyCommand):
|
|||
else:
|
||||
requests = self.requests.get(lvl, [])
|
||||
|
||||
print("# Requests ", "-"*65)
|
||||
print("# Requests ", "-" * 65)
|
||||
display.pprint(requests, colorize=colour)
|
||||
|
||||
def print_results(self, opts):
|
||||
colour = not opts.nocolour
|
||||
|
||||
if opts.verbose:
|
||||
for level in range(1, self.max_level+1):
|
||||
for level in range(1, self.max_level + 1):
|
||||
print('\n>>> DEPTH LEVEL: %s <<<' % level)
|
||||
if not opts.noitems:
|
||||
self.print_items(level, colour)
|
||||
|
|
|
|||
|
|
@ -1,9 +1,9 @@
|
|||
from __future__ import print_function
|
||||
import json
|
||||
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.settings import BaseSettings
|
||||
|
||||
|
||||
class Command(ScrapyCommand):
|
||||
|
||||
requires_project = False
|
||||
|
|
|
|||
|
|
@ -6,8 +6,8 @@ See documentation in docs/topics/shell.rst
|
|||
from threading import Thread
|
||||
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.shell import Shell
|
||||
from scrapy.http import Request
|
||||
from scrapy.shell import Shell
|
||||
from scrapy.utils.spider import spidercls_for_request, DefaultSpider
|
||||
from scrapy.utils.url import guess_scheme
|
||||
|
||||
|
|
|
|||
|
|
@ -1,4 +1,3 @@
|
|||
from __future__ import print_function
|
||||
import re
|
||||
import os
|
||||
import string
|
||||
|
|
@ -44,8 +43,8 @@ class Command(ScrapyCommand):
|
|||
return False
|
||||
|
||||
if not re.search(r'^[_a-zA-Z]\w*$', project_name):
|
||||
print('Error: Project names must begin with a letter and contain'\
|
||||
' only\nletters, numbers and underscores')
|
||||
print('Error: Project names must begin with a letter and contain'
|
||||
' only\nletters, numbers and underscores')
|
||||
elif _module_exists(project_name):
|
||||
print('Error: Module %r already exists' % project_name)
|
||||
else:
|
||||
|
|
@ -119,4 +118,3 @@ class Command(ScrapyCommand):
|
|||
_templates_base_dir = self.settings['TEMPLATES_DIR'] or \
|
||||
join(scrapy.__path__[0], 'templates')
|
||||
return join(_templates_base_dir, 'project')
|
||||
|
||||
|
|
|
|||
|
|
@ -1,5 +1,3 @@
|
|||
from __future__ import print_function
|
||||
|
||||
import scrapy
|
||||
from scrapy.commands import ScrapyCommand
|
||||
from scrapy.utils.versions import scrapy_components_versions
|
||||
|
|
@ -30,4 +28,3 @@ class Command(ScrapyCommand):
|
|||
print(patt % (name, version))
|
||||
else:
|
||||
print("Scrapy %s" % scrapy.__version__)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
from scrapy.commands import fetch, ScrapyCommand
|
||||
from scrapy.commands import fetch
|
||||
from scrapy.utils.response import open_in_browser
|
||||
|
||||
|
||||
class Command(fetch.Command):
|
||||
|
||||
def short_desc(self):
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ from scrapy.item import BaseItem
|
|||
from scrapy.http import Request
|
||||
from scrapy.exceptions import ContractFail
|
||||
|
||||
from . import Contract
|
||||
from scrapy.contracts import Contract
|
||||
|
||||
|
||||
# contracts
|
||||
|
|
@ -86,8 +86,8 @@ class ReturnsContract(Contract):
|
|||
else:
|
||||
expected = '%s..%s' % (self.min_bound, self.max_bound)
|
||||
|
||||
raise ContractFail("Returned %s %s, expected %s" % \
|
||||
(occurrences, self.obj_name, expected))
|
||||
raise ContractFail("Returned %s %s, expected %s" %
|
||||
(occurrences, self.obj_name, expected))
|
||||
|
||||
|
||||
class ScrapesContract(Contract):
|
||||
|
|
|
|||
|
|
@ -1,19 +1,16 @@
|
|||
from __future__ import absolute_import
|
||||
import random
|
||||
import warnings
|
||||
from time import time
|
||||
from datetime import datetime
|
||||
from collections import deque
|
||||
|
||||
import six
|
||||
from twisted.internet import reactor, defer, task
|
||||
|
||||
from scrapy.utils.defer import mustbe_deferred
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
from scrapy.resolver import dnscache
|
||||
from scrapy import signals
|
||||
from .middleware import DownloaderMiddlewareManager
|
||||
from .handlers import DownloadHandlers
|
||||
from scrapy.core.downloader.middleware import DownloaderMiddlewareManager
|
||||
from scrapy.core.downloader.handlers import DownloadHandlers
|
||||
|
||||
|
||||
class Slot(object):
|
||||
|
|
@ -184,13 +181,16 @@ class Downloader(object):
|
|||
def finish_transferring(_):
|
||||
slot.transferring.remove(request)
|
||||
self._process_queue(spider, slot)
|
||||
self.signals.send_catch_log(signal=signals.request_left_downloader,
|
||||
request=request,
|
||||
spider=spider)
|
||||
return _
|
||||
|
||||
return dfd.addBoth(finish_transferring)
|
||||
|
||||
def close(self):
|
||||
self._slot_gc_loop.stop()
|
||||
for slot in six.itervalues(self.slots):
|
||||
for slot in self.slots.values():
|
||||
slot.close()
|
||||
|
||||
def _slot_gc(self, age=60):
|
||||
|
|
|
|||
|
|
@ -67,15 +67,18 @@ class BrowserLikeContextFactory(ScrapyClientContextFactory):
|
|||
"""
|
||||
Twisted-recommended context factory for web clients.
|
||||
|
||||
Quoting https://twistedmatrix.com/documents/current/api/twisted.web.client.Agent.html:
|
||||
"The default is to use a BrowserLikePolicyForHTTPS,
|
||||
so unless you have special requirements you can leave this as-is."
|
||||
Quoting the documentation of the :class:`~twisted.web.client.Agent` class:
|
||||
|
||||
creatorForNetloc() is the same as BrowserLikePolicyForHTTPS
|
||||
except this context factory allows setting the TLS/SSL method to use.
|
||||
The default is to use a
|
||||
:class:`~twisted.web.client.BrowserLikePolicyForHTTPS`, so unless you
|
||||
have special requirements you can leave this as-is.
|
||||
|
||||
Default OpenSSL method is TLS_METHOD (also called SSLv23_METHOD)
|
||||
which allows TLS protocol negotiation.
|
||||
:meth:`creatorForNetloc` is the same as
|
||||
:class:`~twisted.web.client.BrowserLikePolicyForHTTPS` except this context
|
||||
factory allows setting the TLS/SSL method to use.
|
||||
|
||||
The default OpenSSL method is ``TLS_METHOD`` (also called
|
||||
``SSLv23_METHOD``) which allows TLS protocol negotiation.
|
||||
"""
|
||||
def creatorForNetloc(self, hostname, port):
|
||||
|
||||
|
|
|
|||
|
|
@ -1,19 +1,20 @@
|
|||
"""Download handlers for different schemes"""
|
||||
|
||||
import logging
|
||||
|
||||
from twisted.internet import defer
|
||||
import six
|
||||
from scrapy.exceptions import NotSupported, NotConfigured
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.python import without_none_values
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.exceptions import NotConfigured, NotSupported
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.python import without_none_values
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class DownloadHandlers(object):
|
||||
class DownloadHandlers:
|
||||
|
||||
def __init__(self, crawler):
|
||||
self._crawler = crawler
|
||||
|
|
@ -22,7 +23,7 @@ class DownloadHandlers(object):
|
|||
self._notconfigured = {} # remembers failed handlers
|
||||
handlers = without_none_values(
|
||||
crawler.settings.getwithbase('DOWNLOAD_HANDLERS'))
|
||||
for scheme, clspath in six.iteritems(handlers):
|
||||
for scheme, clspath in handlers.items():
|
||||
self._schemes[scheme] = clspath
|
||||
self._load_handler(scheme, skip_lazy=True)
|
||||
|
||||
|
|
@ -48,7 +49,11 @@ class DownloadHandlers(object):
|
|||
dhcls = load_object(path)
|
||||
if skip_lazy and getattr(dhcls, 'lazy', True):
|
||||
return None
|
||||
dh = dhcls(self._crawler.settings)
|
||||
dh = create_instance(
|
||||
objcls=dhcls,
|
||||
settings=self._crawler.settings,
|
||||
crawler=self._crawler,
|
||||
)
|
||||
except NotConfigured as ex:
|
||||
self._notconfigured[scheme] = str(ex)
|
||||
return None
|
||||
|
|
|
|||
|
|
@ -5,20 +5,17 @@ from scrapy.responsetypes import responsetypes
|
|||
from scrapy.utils.decorators import defers
|
||||
|
||||
|
||||
class DataURIDownloadHandler(object):
|
||||
class DataURIDownloadHandler:
|
||||
lazy = False
|
||||
|
||||
def __init__(self, settings):
|
||||
super(DataURIDownloadHandler, self).__init__()
|
||||
|
||||
@defers
|
||||
def download_request(self, request, spider):
|
||||
uri = parse_data_uri(request.url)
|
||||
respcls = responsetypes.from_mimetype(uri.media_type)
|
||||
|
||||
resp_kwargs = {}
|
||||
if (issubclass(respcls, TextResponse) and
|
||||
uri.media_type.split('/')[0] == 'text'):
|
||||
if (issubclass(respcls, TextResponse)
|
||||
and uri.media_type.split('/')[0] == 'text'):
|
||||
charset = uri.media_type_parameters.get('charset')
|
||||
resp_kwargs['encoding'] = charset
|
||||
|
||||
|
|
|
|||
|
|
@ -1,14 +1,12 @@
|
|||
from w3lib.url import file_uri_to_path
|
||||
|
||||
from scrapy.responsetypes import responsetypes
|
||||
from scrapy.utils.decorators import defers
|
||||
|
||||
|
||||
class FileDownloadHandler(object):
|
||||
class FileDownloadHandler:
|
||||
lazy = False
|
||||
|
||||
def __init__(self, settings):
|
||||
pass
|
||||
|
||||
@defers
|
||||
def download_request(self, request, spider):
|
||||
filepath = file_uri_to_path(request.url)
|
||||
|
|
|
|||
|
|
@ -30,11 +30,11 @@ In case of status 200 request, response.headers will come with two keys:
|
|||
|
||||
import re
|
||||
from io import BytesIO
|
||||
from six.moves.urllib.parse import unquote
|
||||
from urllib.parse import unquote
|
||||
|
||||
from twisted.internet import reactor
|
||||
from twisted.protocols.ftp import FTPClient, CommandFailed
|
||||
from twisted.internet.protocol import Protocol, ClientCreator
|
||||
from twisted.internet.protocol import ClientCreator, Protocol
|
||||
from twisted.protocols.ftp import CommandFailed, FTPClient
|
||||
|
||||
from scrapy.http import Response
|
||||
from scrapy.responsetypes import responsetypes
|
||||
|
|
@ -59,10 +59,11 @@ class ReceivedDataProtocol(Protocol):
|
|||
def close(self):
|
||||
self.body.close() if self.filename else self.body.seek(0)
|
||||
|
||||
|
||||
_CODE_RE = re.compile(r"\d+")
|
||||
|
||||
|
||||
class FTPDownloadHandler(object):
|
||||
class FTPDownloadHandler:
|
||||
lazy = False
|
||||
|
||||
CODE_MAPPING = {
|
||||
|
|
@ -75,6 +76,10 @@ class FTPDownloadHandler(object):
|
|||
self.default_password = settings['FTP_PASSWORD']
|
||||
self.passive_mode = settings['FTP_PASSIVE_MODE']
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
return cls(crawler.settings)
|
||||
|
||||
def download_request(self, request, spider):
|
||||
parsed_url = urlparse_cached(request)
|
||||
user = request.meta.get("ftp_user", self.default_user)
|
||||
|
|
@ -112,4 +117,3 @@ class FTPDownloadHandler(object):
|
|||
httpcode = self.CODE_MAPPING.get(ftpcode, self.CODE_MAPPING["default"])
|
||||
return Response(url=request.url, status=httpcode, body=to_bytes(message))
|
||||
raise result.type(result.value)
|
||||
|
||||
|
|
|
|||
|
|
@ -1,3 +1,4 @@
|
|||
from __future__ import absolute_import
|
||||
from .http10 import HTTP10DownloadHandler
|
||||
from .http11 import HTTP11DownloadHandler as HTTPDownloadHandler
|
||||
from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler
|
||||
from scrapy.core.downloader.handlers.http11 import (
|
||||
HTTP11DownloadHandler as HTTPDownloadHandler,
|
||||
)
|
||||
|
|
|
|||
|
|
@ -1,17 +1,23 @@
|
|||
"""Download handlers for http and https schemes
|
||||
"""
|
||||
from twisted.internet import reactor
|
||||
from scrapy.utils.misc import load_object, create_instance
|
||||
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.python import to_unicode
|
||||
|
||||
|
||||
class HTTP10DownloadHandler(object):
|
||||
class HTTP10DownloadHandler:
|
||||
lazy = False
|
||||
|
||||
def __init__(self, settings):
|
||||
def __init__(self, settings, crawler=None):
|
||||
self.HTTPClientFactory = load_object(settings['DOWNLOADER_HTTPCLIENTFACTORY'])
|
||||
self.ClientContextFactory = load_object(settings['DOWNLOADER_CLIENTCONTEXTFACTORY'])
|
||||
self._settings = settings
|
||||
self._crawler = crawler
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
return cls(crawler.settings, crawler)
|
||||
|
||||
def download_request(self, request, spider):
|
||||
"""Return a deferred for the HTTP download"""
|
||||
|
|
@ -22,7 +28,11 @@ class HTTP10DownloadHandler(object):
|
|||
def _connect(self, factory):
|
||||
host, port = to_unicode(factory.host), factory.port
|
||||
if factory.scheme == b'https':
|
||||
client_context_factory = create_instance(self.ClientContextFactory, settings=self._settings, crawler=None)
|
||||
client_context_factory = create_instance(
|
||||
objcls=self.ClientContextFactory,
|
||||
settings=self._settings,
|
||||
crawler=self._crawler,
|
||||
)
|
||||
return reactor.connectSSL(host, port, factory, client_context_factory)
|
||||
else:
|
||||
return reactor.connectTCP(host, port, factory)
|
||||
|
|
|
|||
|
|
@ -1,36 +1,38 @@
|
|||
"""Download handlers for http and https schemes"""
|
||||
|
||||
import re
|
||||
import logging
|
||||
import re
|
||||
import warnings
|
||||
from contextlib import suppress
|
||||
from io import BytesIO
|
||||
from time import time
|
||||
import warnings
|
||||
from six.moves.urllib.parse import urldefrag
|
||||
from urllib.parse import urldefrag
|
||||
|
||||
from zope.interface import implementer
|
||||
from twisted.internet import defer, reactor, protocol
|
||||
from twisted.internet import defer, protocol, reactor, ssl
|
||||
from twisted.internet.endpoints import TCP4ClientEndpoint
|
||||
from twisted.internet.error import TimeoutError
|
||||
from twisted.web.client import Agent, HTTPConnectionPool, ResponseDone, ResponseFailed, URI
|
||||
from twisted.web.http import _DataLoss, PotentialDataLoss
|
||||
from twisted.web.http_headers import Headers as TxHeaders
|
||||
from twisted.web.iweb import IBodyProducer, UNKNOWN_LENGTH
|
||||
from twisted.internet.error import TimeoutError
|
||||
from twisted.web.http import _DataLoss, PotentialDataLoss
|
||||
from twisted.web.client import Agent, ResponseDone, HTTPConnectionPool, ResponseFailed, URI
|
||||
from twisted.internet.endpoints import TCP4ClientEndpoint
|
||||
from zope.interface import implementer
|
||||
|
||||
from scrapy.core.downloader.tls import openssl_methods
|
||||
from scrapy.core.downloader.webclient import _parse
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import Headers
|
||||
from scrapy.responsetypes import responsetypes
|
||||
from scrapy.core.downloader.webclient import _parse
|
||||
from scrapy.core.downloader.tls import openssl_methods
|
||||
from scrapy.utils.misc import load_object, create_instance
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.python import to_bytes, to_unicode
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class HTTP11DownloadHandler(object):
|
||||
class HTTP11DownloadHandler:
|
||||
lazy = False
|
||||
|
||||
def __init__(self, settings):
|
||||
def __init__(self, settings, crawler=None):
|
||||
self._pool = HTTPConnectionPool(reactor, persistent=True)
|
||||
self._pool.maxPersistentPerHost = settings.getint('CONCURRENT_REQUESTS_PER_DOMAIN')
|
||||
self._pool._factory.noisy = False
|
||||
|
|
@ -40,17 +42,17 @@ class HTTP11DownloadHandler(object):
|
|||
# try method-aware context factory
|
||||
try:
|
||||
self._contextFactory = create_instance(
|
||||
self._contextFactoryClass,
|
||||
objcls=self._contextFactoryClass,
|
||||
settings=settings,
|
||||
crawler=None,
|
||||
crawler=crawler,
|
||||
method=self._sslMethod,
|
||||
)
|
||||
except TypeError:
|
||||
# use context factory defaults
|
||||
self._contextFactory = create_instance(
|
||||
self._contextFactoryClass,
|
||||
objcls=self._contextFactoryClass,
|
||||
settings=settings,
|
||||
crawler=None,
|
||||
crawler=crawler,
|
||||
)
|
||||
msg = """
|
||||
'%s' does not accept `method` argument (type OpenSSL.SSL method,\
|
||||
|
|
@ -63,6 +65,10 @@ class HTTP11DownloadHandler(object):
|
|||
self._fail_on_dataloss = settings.getbool('DOWNLOAD_FAIL_ON_DATALOSS')
|
||||
self._disconnect_timeout = 1
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler):
|
||||
return cls(crawler.settings, crawler)
|
||||
|
||||
def download_request(self, request, spider):
|
||||
"""Return a deferred for the HTTP download"""
|
||||
agent = ScrapyAgent(
|
||||
|
|
@ -174,7 +180,7 @@ def tunnel_request_data(host, port, proxy_auth_header=None):
|
|||
r"""
|
||||
Return binary content of a CONNECT request.
|
||||
|
||||
>>> from scrapy.utils.python import to_native_str as s
|
||||
>>> from scrapy.utils.python import to_unicode as s
|
||||
>>> s(tunnel_request_data("example.com", 8080))
|
||||
'CONNECT example.com:8080 HTTP/1.1\r\nHost: example.com:8080\r\n\r\n'
|
||||
>>> s(tunnel_request_data("example.com", 8080, b"123"))
|
||||
|
|
@ -285,6 +291,12 @@ class ScrapyAgent(object):
|
|||
scheme = _parse(request.url)[0]
|
||||
proxyHost = to_unicode(proxyHost)
|
||||
omitConnectTunnel = b'noconnect' in proxyParams
|
||||
if omitConnectTunnel:
|
||||
warnings.warn("Using HTTPS proxies in the noconnect mode is deprecated. "
|
||||
"If you use Crawlera, it doesn't require this mode anymore, "
|
||||
"so you should update scrapy-crawlera to 1.3.0+ "
|
||||
"and remove '?noconnect' from the Crawlera URL.",
|
||||
ScrapyDeprecationWarning)
|
||||
if scheme == b'https' and not omitConnectTunnel:
|
||||
proxyAuth = request.headers.get(b'Proxy-Authorization', None)
|
||||
proxyConf = (proxyHost, proxyPort, proxyAuth)
|
||||
|
|
@ -371,7 +383,7 @@ class ScrapyAgent(object):
|
|||
def _cb_bodyready(self, txresponse, request):
|
||||
# deliverBody hangs for responses without body
|
||||
if txresponse.length == 0:
|
||||
return txresponse, b'', None
|
||||
return txresponse, b'', None, None
|
||||
|
||||
maxsize = request.meta.get('download_maxsize', self._maxsize)
|
||||
warnsize = request.meta.get('download_warnsize', self._warnsize)
|
||||
|
|
@ -407,11 +419,12 @@ class ScrapyAgent(object):
|
|||
return d
|
||||
|
||||
def _cb_bodydone(self, result, request, url):
|
||||
txresponse, body, flags = result
|
||||
txresponse, body, flags, certificate = result
|
||||
status = int(txresponse.code)
|
||||
headers = Headers(txresponse.headers.getAllRawHeaders())
|
||||
respcls = responsetypes.from_args(headers=headers, url=url, body=body)
|
||||
return respcls(url=url, status=status, headers=headers, body=body, flags=flags)
|
||||
return respcls(url=url, status=status, headers=headers, body=body,
|
||||
flags=flags, certificate=certificate)
|
||||
|
||||
|
||||
@implementer(IBodyProducer)
|
||||
|
|
@ -445,6 +458,12 @@ class _ResponseReader(protocol.Protocol):
|
|||
self._fail_on_dataloss_warned = False
|
||||
self._reached_warnsize = False
|
||||
self._bytes_received = 0
|
||||
self._certificate = None
|
||||
|
||||
def connectionMade(self):
|
||||
if self._certificate is None:
|
||||
with suppress(AttributeError):
|
||||
self._certificate = ssl.Certificate(self.transport._producer.getPeerCertificate())
|
||||
|
||||
def dataReceived(self, bodyBytes):
|
||||
# This maybe called several times after cancel was called with buffered data.
|
||||
|
|
@ -477,16 +496,16 @@ class _ResponseReader(protocol.Protocol):
|
|||
|
||||
body = self._bodybuf.getvalue()
|
||||
if reason.check(ResponseDone):
|
||||
self._finished.callback((self._txresponse, body, None))
|
||||
self._finished.callback((self._txresponse, body, None, self._certificate))
|
||||
return
|
||||
|
||||
if reason.check(PotentialDataLoss):
|
||||
self._finished.callback((self._txresponse, body, ['partial']))
|
||||
self._finished.callback((self._txresponse, body, ['partial'], self._certificate))
|
||||
return
|
||||
|
||||
if reason.check(ResponseFailed) and any(r.check(_DataLoss) for r in reason.value.reasons):
|
||||
if not self._fail_on_dataloss:
|
||||
self._finished.callback((self._txresponse, body, ['dataloss']))
|
||||
self._finished.callback((self._txresponse, body, ['dataloss'], self._certificate))
|
||||
return
|
||||
|
||||
elif not self._fail_on_dataloss_warned:
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
from six.moves.urllib.parse import unquote
|
||||
from urllib.parse import unquote
|
||||
|
||||
from scrapy.core.downloader.handlers.http import HTTPDownloadHandler
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
from scrapy.utils.boto import is_botocore
|
||||
from .http import HTTPDownloadHandler
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
from scrapy.utils.misc import create_instance
|
||||
|
||||
|
||||
def _get_boto_connection():
|
||||
|
|
@ -21,7 +22,7 @@ def _get_boto_connection():
|
|||
return http_request.headers
|
||||
|
||||
try:
|
||||
import boto.auth
|
||||
import boto.auth # noqa: F401
|
||||
except ImportError:
|
||||
_S3Connection = _v19_S3Connection
|
||||
else:
|
||||
|
|
@ -30,11 +31,12 @@ def _get_boto_connection():
|
|||
return _S3Connection
|
||||
|
||||
|
||||
class S3DownloadHandler(object):
|
||||
|
||||
def __init__(self, settings, aws_access_key_id=None, aws_secret_access_key=None, \
|
||||
httpdownloadhandler=HTTPDownloadHandler, **kw):
|
||||
class S3DownloadHandler:
|
||||
|
||||
def __init__(self, settings, *,
|
||||
crawler=None,
|
||||
aws_access_key_id=None, aws_secret_access_key=None,
|
||||
httpdownloadhandler=HTTPDownloadHandler, **kw):
|
||||
if not aws_access_key_id:
|
||||
aws_access_key_id = settings['AWS_ACCESS_KEY_ID']
|
||||
if not aws_secret_access_key:
|
||||
|
|
@ -67,7 +69,16 @@ class S3DownloadHandler(object):
|
|||
except Exception as ex:
|
||||
raise NotConfigured(str(ex))
|
||||
|
||||
self._download_http = httpdownloadhandler(settings).download_request
|
||||
_http_handler = create_instance(
|
||||
objcls=httpdownloadhandler,
|
||||
settings=settings,
|
||||
crawler=crawler,
|
||||
)
|
||||
self._download_http = _http_handler.download_request
|
||||
|
||||
@classmethod
|
||||
def from_crawler(cls, crawler, **kwargs):
|
||||
return cls(crawler.settings, crawler=crawler, **kwargs)
|
||||
|
||||
def download_request(self, request, spider):
|
||||
p = urlparse_cached(request)
|
||||
|
|
|
|||
|
|
@ -3,14 +3,12 @@ Downloader Middleware manager
|
|||
|
||||
See documentation in docs/topics/downloader-middleware.rst
|
||||
"""
|
||||
import six
|
||||
|
||||
from twisted.internet import defer
|
||||
|
||||
from scrapy.exceptions import _InvalidOutput
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.middleware import MiddlewareManager
|
||||
from scrapy.utils.defer import mustbe_deferred
|
||||
from scrapy.utils.defer import mustbe_deferred, deferred_from_coro
|
||||
from scrapy.utils.conf import build_component_list
|
||||
|
||||
|
||||
|
|
@ -35,10 +33,10 @@ class DownloaderMiddlewareManager(MiddlewareManager):
|
|||
@defer.inlineCallbacks
|
||||
def process_request(request):
|
||||
for method in self.methods['process_request']:
|
||||
response = yield method(request=request, spider=spider)
|
||||
response = yield deferred_from_coro(method(request=request, spider=spider))
|
||||
if response is not None and not isinstance(response, (Response, Request)):
|
||||
raise _InvalidOutput('Middleware %s.process_request must return None, Response or Request, got %s' % \
|
||||
(six.get_method_self(method).__class__.__name__, response.__class__.__name__))
|
||||
(method.__self__.__class__.__name__, response.__class__.__name__))
|
||||
if response:
|
||||
defer.returnValue(response)
|
||||
defer.returnValue((yield download_func(request=request, spider=spider)))
|
||||
|
|
@ -50,10 +48,10 @@ class DownloaderMiddlewareManager(MiddlewareManager):
|
|||
defer.returnValue(response)
|
||||
|
||||
for method in self.methods['process_response']:
|
||||
response = yield method(request=request, response=response, spider=spider)
|
||||
response = yield deferred_from_coro(method(request=request, response=response, spider=spider))
|
||||
if not isinstance(response, (Response, Request)):
|
||||
raise _InvalidOutput('Middleware %s.process_response must return Response or Request, got %s' % \
|
||||
(six.get_method_self(method).__class__.__name__, type(response)))
|
||||
(method.__self__.__class__.__name__, type(response)))
|
||||
if isinstance(response, Request):
|
||||
defer.returnValue(response)
|
||||
defer.returnValue(response)
|
||||
|
|
@ -62,10 +60,10 @@ class DownloaderMiddlewareManager(MiddlewareManager):
|
|||
def process_exception(_failure):
|
||||
exception = _failure.value
|
||||
for method in self.methods['process_exception']:
|
||||
response = yield method(request=request, exception=exception, spider=spider)
|
||||
response = yield deferred_from_coro(method(request=request, exception=exception, spider=spider))
|
||||
if response is not None and not isinstance(response, (Response, Request)):
|
||||
raise _InvalidOutput('Middleware %s.process_exception must return None, Response or Request, got %s' % \
|
||||
(six.get_method_self(method).__class__.__name__, type(response)))
|
||||
(method.__self__.__class__.__name__, type(response)))
|
||||
if response:
|
||||
defer.returnValue(response)
|
||||
defer.returnValue(_failure)
|
||||
|
|
|
|||
|
|
@ -89,4 +89,5 @@ class ScrapyClientTLSOptions(ClientTLSOptions):
|
|||
'from host "{}" (exception: {})'.format(
|
||||
self._hostnameASCII, repr(e)))
|
||||
|
||||
|
||||
DEFAULT_CIPHERS = AcceptableCiphers.fromOpenSSLCipherString('DEFAULT')
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
from time import time
|
||||
from six.moves.urllib.parse import urlparse, urlunparse, urldefrag
|
||||
from urllib.parse import urlparse, urlunparse, urldefrag
|
||||
|
||||
from twisted.web.client import HTTPClientFactory
|
||||
from twisted.web.http import HTTPClient
|
||||
|
|
@ -42,7 +42,7 @@ class ScrapyHTTPPageGetter(HTTPClient):
|
|||
delimiter = b'\n'
|
||||
|
||||
def connectionMade(self):
|
||||
self.headers = Headers() # bucket for response headers
|
||||
self.headers = Headers() # bucket for response headers
|
||||
|
||||
# Method command
|
||||
self.sendCommand(self.factory.method, self.factory.path)
|
||||
|
|
@ -88,9 +88,9 @@ class ScrapyHTTPPageGetter(HTTPClient):
|
|||
if self.factory.url.startswith(b'https'):
|
||||
self.transport.stopProducing()
|
||||
|
||||
self.factory.noPage(\
|
||||
defer.TimeoutError("Getting %s took longer than %s seconds." % \
|
||||
(self.factory.url, self.factory.timeout)))
|
||||
self.factory.noPage(
|
||||
defer.TimeoutError("Getting %s took longer than %s seconds." %
|
||||
(self.factory.url, self.factory.timeout)))
|
||||
|
||||
|
||||
class ScrapyHTTPClientFactory(HTTPClientFactory):
|
||||
|
|
@ -140,7 +140,7 @@ class ScrapyHTTPClientFactory(HTTPClientFactory):
|
|||
self.headers['Content-Length'] = 0
|
||||
|
||||
def _build_response(self, body, request):
|
||||
request.meta['download_latency'] = self.headers_time-self.start_time
|
||||
request.meta['download_latency'] = self.headers_time - self.start_time
|
||||
status = int(self.status)
|
||||
headers = Headers(self.response_headers)
|
||||
respcls = responsetypes.from_args(headers=headers, url=self._url)
|
||||
|
|
@ -157,4 +157,3 @@ class ScrapyHTTPClientFactory(HTTPClientFactory):
|
|||
def gotHeaders(self, headers):
|
||||
self.headers_time = time()
|
||||
self.response_headers = headers
|
||||
|
||||
|
|
|
|||
|
|
@ -25,7 +25,7 @@ class Slot(object):
|
|||
|
||||
def __init__(self, start_requests, close_if_idle, nextcall, scheduler):
|
||||
self.closing = False
|
||||
self.inprogress = set() # requests in progress
|
||||
self.inprogress = set() # requests in progress
|
||||
self.start_requests = iter(start_requests)
|
||||
self.close_if_idle = close_if_idle
|
||||
self.nextcall = nextcall
|
||||
|
|
@ -230,6 +230,7 @@ class ExecutionEngine(object):
|
|||
def _download(self, request, spider):
|
||||
slot = self.slot
|
||||
slot.add_request(request)
|
||||
|
||||
def _on_success(response):
|
||||
assert isinstance(response, (Response, Request))
|
||||
if isinstance(response, Response):
|
||||
|
|
|
|||
|
|
@ -119,7 +119,7 @@ class Scheduler(object):
|
|||
if self.dqs is None:
|
||||
return
|
||||
try:
|
||||
self.dqs.push(request, -request.priority)
|
||||
self.dqs.push(request)
|
||||
except ValueError as e: # non serializable request
|
||||
if self.logunser:
|
||||
msg = ("Unable to serialize request: %(request)s - reason:"
|
||||
|
|
@ -135,35 +135,29 @@ class Scheduler(object):
|
|||
return True
|
||||
|
||||
def _mqpush(self, request):
|
||||
self.mqs.push(request, -request.priority)
|
||||
self.mqs.push(request)
|
||||
|
||||
def _dqpop(self):
|
||||
if self.dqs:
|
||||
return self.dqs.pop()
|
||||
|
||||
def _newmq(self, priority):
|
||||
""" Factory for creating memory queues. """
|
||||
return self.mqclass()
|
||||
|
||||
def _newdq(self, priority):
|
||||
""" Factory for creating disk queues. """
|
||||
path = join(self.dqdir, 'p%s' % (priority, ))
|
||||
return self.dqclass(path)
|
||||
|
||||
def _mq(self):
|
||||
""" Create a new priority queue instance, with in-memory storage """
|
||||
return create_instance(self.pqclass, None, self.crawler, self._newmq,
|
||||
serialize=False)
|
||||
return create_instance(self.pqclass,
|
||||
settings=None,
|
||||
crawler=self.crawler,
|
||||
downstream_queue_cls=self.mqclass,
|
||||
key='')
|
||||
|
||||
def _dq(self):
|
||||
""" Create a new priority queue instance, with disk storage """
|
||||
state = self._read_dqs_state(self.dqdir)
|
||||
q = create_instance(self.pqclass,
|
||||
None,
|
||||
self.crawler,
|
||||
self._newdq,
|
||||
state,
|
||||
serialize=True)
|
||||
settings=None,
|
||||
crawler=self.crawler,
|
||||
downstream_queue_cls=self.dqclass,
|
||||
key=self.dqdir,
|
||||
startprios=state)
|
||||
if q:
|
||||
logger.info("Resuming crawl (%(queuesize)d requests scheduled)",
|
||||
{'queuesize': len(q)}, extra={'spider': self.spider})
|
||||
|
|
|
|||
|
|
@ -9,14 +9,14 @@ from twisted.internet import defer
|
|||
|
||||
from scrapy.utils.defer import defer_result, defer_succeed, parallel, iter_errback
|
||||
from scrapy.utils.spider import iterate_spider_output
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.misc import load_object, warn_on_generator_with_return_value
|
||||
from scrapy.utils.log import logformatter_adapter, failure_to_exc_info
|
||||
from scrapy.exceptions import CloseSpider, DropItem, IgnoreRequest
|
||||
from scrapy import signals
|
||||
from scrapy.http import Request, Response
|
||||
from scrapy.item import BaseItem
|
||||
from scrapy.core.spidermw import SpiderMiddlewareManager
|
||||
from scrapy.utils.request import referer_str
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
|
@ -77,7 +77,7 @@ class Scraper(object):
|
|||
@defer.inlineCallbacks
|
||||
def open_spider(self, spider):
|
||||
"""Open the given spider for scraping and allocate resources for it"""
|
||||
self.slot = Slot()
|
||||
self.slot = Slot(self.crawler.settings.getint('SCRAPER_SLOT_MAX_ACTIVE_SIZE'))
|
||||
yield self.itemproc.open_spider(spider)
|
||||
|
||||
def close_spider(self, spider):
|
||||
|
|
@ -99,11 +99,13 @@ class Scraper(object):
|
|||
def enqueue_scrape(self, response, request, spider):
|
||||
slot = self.slot
|
||||
dfd = slot.add_response_request(response, request)
|
||||
|
||||
def finish_scraping(_):
|
||||
slot.finish_response(response, request)
|
||||
self._check_if_closing(spider, slot)
|
||||
self._scrape_next(spider, slot)
|
||||
return _
|
||||
|
||||
dfd.addBoth(finish_scraping)
|
||||
dfd.addErrback(
|
||||
lambda f: logger.error('Scraper bug processing %(request)s',
|
||||
|
|
@ -123,7 +125,7 @@ class Scraper(object):
|
|||
callback/errback"""
|
||||
assert isinstance(response, (Response, Failure))
|
||||
|
||||
dfd = self._scrape2(response, request, spider) # returns spiders processed output
|
||||
dfd = self._scrape2(response, request, spider) # returns spider's processed output
|
||||
dfd.addErrback(self.handle_spider_error, request, response, spider)
|
||||
dfd.addCallback(self.handle_spider_output, request, response, spider)
|
||||
return dfd
|
||||
|
|
@ -142,7 +144,10 @@ class Scraper(object):
|
|||
def call_spider(self, result, request, spider):
|
||||
result.request = request
|
||||
dfd = defer_result(result)
|
||||
dfd.addCallbacks(callback=request.callback or spider.parse,
|
||||
callback = request.callback or spider.parse
|
||||
warn_on_generator_with_return_value(spider, callback)
|
||||
warn_on_generator_with_return_value(spider, request.errback)
|
||||
dfd.addCallbacks(callback=callback,
|
||||
errback=request.errback,
|
||||
callbackKeywords=request.cb_kwargs)
|
||||
return dfd.addCallback(iterate_spider_output)
|
||||
|
|
@ -152,9 +157,9 @@ class Scraper(object):
|
|||
if isinstance(exc, CloseSpider):
|
||||
self.crawler.engine.close_spider(spider, exc.reason or 'cancelled')
|
||||
return
|
||||
logger.error(
|
||||
"Spider error processing %(request)s (referer: %(referer)s)",
|
||||
{'request': request, 'referer': referer_str(request)},
|
||||
logkws = self.logformatter.spider_error(_failure, request, response, spider)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
exc_info=failure_to_exc_info(_failure),
|
||||
extra={'spider': spider}
|
||||
)
|
||||
|
|
@ -172,8 +177,8 @@ class Scraper(object):
|
|||
if not result:
|
||||
return defer_succeed(None)
|
||||
it = iter_errback(result, self.handle_spider_error, request, response, spider)
|
||||
dfd = parallel(it, self.concurrent_items,
|
||||
self._process_spidermw_output, request, response, spider)
|
||||
dfd = parallel(it, self.concurrent_items, self._process_spidermw_output,
|
||||
request, response, spider)
|
||||
return dfd
|
||||
|
||||
def _process_spidermw_output(self, output, request, response, spider):
|
||||
|
|
@ -200,19 +205,23 @@ class Scraper(object):
|
|||
"""Log and silence errors that come from the engine (typically download
|
||||
errors that got propagated thru here)
|
||||
"""
|
||||
if (isinstance(download_failure, Failure) and
|
||||
not download_failure.check(IgnoreRequest)):
|
||||
if isinstance(download_failure, Failure) and not download_failure.check(IgnoreRequest):
|
||||
if download_failure.frames:
|
||||
logger.error('Error downloading %(request)s',
|
||||
{'request': request},
|
||||
exc_info=failure_to_exc_info(download_failure),
|
||||
extra={'spider': spider})
|
||||
logkws = self.logformatter.download_error(download_failure, request, spider)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
extra={'spider': spider},
|
||||
exc_info=failure_to_exc_info(download_failure),
|
||||
)
|
||||
else:
|
||||
errmsg = download_failure.getErrorMessage()
|
||||
if errmsg:
|
||||
logger.error('Error downloading %(request)s: %(errmsg)s',
|
||||
{'request': request, 'errmsg': errmsg},
|
||||
extra={'spider': spider})
|
||||
logkws = self.logformatter.download_error(
|
||||
download_failure, request, spider, errmsg)
|
||||
logger.log(
|
||||
*logformatter_adapter(logkws),
|
||||
extra={'spider': spider},
|
||||
)
|
||||
|
||||
if spider_failure is not download_failure:
|
||||
return spider_failure
|
||||
|
|
@ -231,9 +240,9 @@ class Scraper(object):
|
|||
signal=signals.item_dropped, item=item, response=response,
|
||||
spider=spider, exception=output.value)
|
||||
else:
|
||||
logger.error('Error processing %(item)s', {'item': item},
|
||||
exc_info=failure_to_exc_info(output),
|
||||
extra={'spider': spider})
|
||||
logkws = self.logformatter.item_error(item, ex, response, spider)
|
||||
logger.log(*logformatter_adapter(logkws), extra={'spider': spider},
|
||||
exc_info=failure_to_exc_info(output))
|
||||
return self.signals.send_catch_log_deferred(
|
||||
signal=signals.item_error, item=item, response=response,
|
||||
spider=spider, failure=output)
|
||||
|
|
@ -244,4 +253,3 @@ class Scraper(object):
|
|||
return self.signals.send_catch_log_deferred(
|
||||
signal=signals.item_scraped, item=output, response=response,
|
||||
spider=spider)
|
||||
|
||||
|
|
|
|||
|
|
@ -3,14 +3,14 @@ Spider Middleware manager
|
|||
|
||||
See documentation in docs/topics/spider-middleware.rst
|
||||
"""
|
||||
from itertools import chain, islice
|
||||
from itertools import islice
|
||||
|
||||
import six
|
||||
from twisted.python.failure import Failure
|
||||
|
||||
from scrapy.exceptions import _InvalidOutput
|
||||
from scrapy.middleware import MiddlewareManager
|
||||
from scrapy.utils.defer import mustbe_deferred
|
||||
from scrapy.utils.conf import build_component_list
|
||||
from scrapy.utils.defer import mustbe_deferred
|
||||
from scrapy.utils.python import MutableChain
|
||||
|
||||
|
||||
|
|
@ -18,6 +18,13 @@ def _isiterable(possible_iterator):
|
|||
return hasattr(possible_iterator, '__iter__')
|
||||
|
||||
|
||||
def _fname(f):
|
||||
return "%s.%s".format(
|
||||
f.__self__.__class__.__name__,
|
||||
f.__func__.__name__
|
||||
)
|
||||
|
||||
|
||||
class SpiderMiddlewareManager(MiddlewareManager):
|
||||
|
||||
component_name = 'spider middleware'
|
||||
|
|
@ -32,27 +39,36 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
self.methods['process_spider_input'].append(mw.process_spider_input)
|
||||
if hasattr(mw, 'process_start_requests'):
|
||||
self.methods['process_start_requests'].appendleft(mw.process_start_requests)
|
||||
self.methods['process_spider_output'].appendleft(getattr(mw, 'process_spider_output', None))
|
||||
self.methods['process_spider_exception'].appendleft(getattr(mw, 'process_spider_exception', None))
|
||||
process_spider_output = getattr(mw, 'process_spider_output', None)
|
||||
self.methods['process_spider_output'].appendleft(process_spider_output)
|
||||
process_spider_exception = getattr(mw, 'process_spider_exception', None)
|
||||
self.methods['process_spider_exception'].appendleft(process_spider_exception)
|
||||
|
||||
def scrape_response(self, scrape_func, response, request, spider):
|
||||
fname = lambda f:'%s.%s' % (
|
||||
six.get_method_self(f).__class__.__name__,
|
||||
six.get_method_function(f).__name__)
|
||||
|
||||
def process_spider_input(response):
|
||||
for method in self.methods['process_spider_input']:
|
||||
try:
|
||||
result = method(response=response, spider=spider)
|
||||
if result is not None:
|
||||
raise _InvalidOutput('Middleware {} must return None or raise an exception, got {}' \
|
||||
.format(fname(method), type(result)))
|
||||
msg = "Middleware {} must return None or raise an exception, got {}"
|
||||
raise _InvalidOutput(msg.format(_fname(method), type(result)))
|
||||
except _InvalidOutput:
|
||||
raise
|
||||
except Exception:
|
||||
return scrape_func(Failure(), request, spider)
|
||||
return scrape_func(response, request, spider)
|
||||
|
||||
def _evaluate_iterable(iterable, exception_processor_index, recover_to):
|
||||
try:
|
||||
for r in iterable:
|
||||
yield r
|
||||
except Exception as ex:
|
||||
exception_result = process_spider_exception(Failure(ex), exception_processor_index)
|
||||
if isinstance(exception_result, Failure):
|
||||
raise
|
||||
recover_to.extend(exception_result)
|
||||
|
||||
def process_spider_exception(_failure, start_index=0):
|
||||
exception = _failure.value
|
||||
# don't handle _InvalidOutput exception
|
||||
|
|
@ -66,12 +82,12 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
if _isiterable(result):
|
||||
# stop exception handling by handing control over to the
|
||||
# process_spider_output chain if an iterable has been returned
|
||||
return process_spider_output(result, method_index+1)
|
||||
return process_spider_output(result, method_index + 1)
|
||||
elif result is None:
|
||||
continue
|
||||
else:
|
||||
raise _InvalidOutput('Middleware {} must return None or an iterable, got {}' \
|
||||
.format(fname(method), type(result)))
|
||||
msg = "Middleware {} must return None or an iterable, got {}"
|
||||
raise _InvalidOutput(msg.format(_fname(method), type(result)))
|
||||
return _failure
|
||||
|
||||
def process_spider_output(result, start_index=0):
|
||||
|
|
@ -79,38 +95,33 @@ class SpiderMiddlewareManager(MiddlewareManager):
|
|||
# chain, they went through it already from the process_spider_exception method
|
||||
recovered = MutableChain()
|
||||
|
||||
def evaluate_iterable(iterable, index):
|
||||
try:
|
||||
for r in iterable:
|
||||
yield r
|
||||
except Exception as ex:
|
||||
exception_result = process_spider_exception(Failure(ex), index+1)
|
||||
if isinstance(exception_result, Failure):
|
||||
raise
|
||||
recovered.extend(exception_result)
|
||||
|
||||
method_list = islice(self.methods['process_spider_output'], start_index, None)
|
||||
for method_index, method in enumerate(method_list, start=start_index):
|
||||
if method is None:
|
||||
continue
|
||||
# the following might fail directly if the output value is not a generator
|
||||
try:
|
||||
# might fail directly if the output value is not a generator
|
||||
result = method(response=response, result=result, spider=spider)
|
||||
except Exception as ex:
|
||||
exception_result = process_spider_exception(Failure(ex), method_index+1)
|
||||
exception_result = process_spider_exception(Failure(ex), method_index + 1)
|
||||
if isinstance(exception_result, Failure):
|
||||
raise
|
||||
return exception_result
|
||||
if _isiterable(result):
|
||||
result = evaluate_iterable(result, method_index)
|
||||
result = _evaluate_iterable(result, method_index + 1, recovered)
|
||||
else:
|
||||
raise _InvalidOutput('Middleware {} must return an iterable, got {}' \
|
||||
.format(fname(method), type(result)))
|
||||
msg = "Middleware {} must return an iterable, got {}"
|
||||
raise _InvalidOutput(msg.format(_fname(method), type(result)))
|
||||
|
||||
return chain(result, recovered)
|
||||
return MutableChain(result, recovered)
|
||||
|
||||
def process_callback_output(result):
|
||||
recovered = MutableChain()
|
||||
result = _evaluate_iterable(result, 0, recovered)
|
||||
return MutableChain(process_spider_output(result), recovered)
|
||||
|
||||
dfd = mustbe_deferred(process_spider_input, response)
|
||||
dfd.addCallbacks(callback=process_spider_output, errback=process_spider_exception)
|
||||
dfd.addCallbacks(callback=process_callback_output, errback=process_spider_exception)
|
||||
return dfd
|
||||
|
||||
def process_start_requests(self, start_requests, spider):
|
||||
|
|
|
|||
|
|
@ -1,36 +1,38 @@
|
|||
import six
|
||||
import signal
|
||||
import logging
|
||||
import pprint
|
||||
import signal
|
||||
import warnings
|
||||
|
||||
import sys
|
||||
from twisted.internet import reactor, defer
|
||||
from zope.interface.verify import verifyClass, DoesNotImplement
|
||||
from twisted.internet import defer
|
||||
from zope.interface.verify import DoesNotImplement, verifyClass
|
||||
|
||||
from scrapy import Spider
|
||||
from scrapy import signals, Spider
|
||||
from scrapy.core.engine import ExecutionEngine
|
||||
from scrapy.resolver import CachingThreadedResolver
|
||||
from scrapy.interfaces import ISpiderLoader
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.extension import ExtensionManager
|
||||
from scrapy.interfaces import ISpiderLoader
|
||||
from scrapy.settings import overridden_settings, Settings
|
||||
from scrapy.signalmanager import SignalManager
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.utils.ossignal import install_shutdown_handlers, signal_names
|
||||
from scrapy.utils.misc import load_object
|
||||
from scrapy.utils.log import (
|
||||
LogCounterHandler, configure_logging, log_scrapy_info,
|
||||
get_scrapy_root_handler, install_scrapy_root_handler)
|
||||
from scrapy import signals
|
||||
configure_logging,
|
||||
get_scrapy_root_handler,
|
||||
install_scrapy_root_handler,
|
||||
log_scrapy_info,
|
||||
LogCounterHandler,
|
||||
)
|
||||
from scrapy.utils.misc import create_instance, load_object
|
||||
from scrapy.utils.ossignal import install_shutdown_handlers, signal_names
|
||||
from scrapy.utils.reactor import install_reactor, verify_installed_reactor
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
class Crawler(object):
|
||||
class Crawler:
|
||||
|
||||
def __init__(self, spidercls, settings=None):
|
||||
if isinstance(spidercls, Spider):
|
||||
raise ValueError(
|
||||
'The spidercls argument must be a class, not an object')
|
||||
raise ValueError('The spidercls argument must be a class, not an object')
|
||||
|
||||
if isinstance(settings, dict) or settings is None:
|
||||
settings = Settings(settings)
|
||||
|
|
@ -46,7 +48,8 @@ class Crawler(object):
|
|||
logging.root.addHandler(handler)
|
||||
|
||||
d = dict(overridden_settings(self.settings))
|
||||
logger.info("Overridden settings: %(settings)r", {'settings': d})
|
||||
logger.info("Overridden settings:\n%(settings)s",
|
||||
{'settings': pprint.pformat(d)})
|
||||
|
||||
if get_scrapy_root_handler() is not None:
|
||||
# scrapy root handler already installed: update it with new settings
|
||||
|
|
@ -88,20 +91,9 @@ class Crawler(object):
|
|||
yield self.engine.open_spider(self.spider, start_requests)
|
||||
yield defer.maybeDeferred(self.engine.start)
|
||||
except Exception:
|
||||
# In Python 2 reraising an exception after yield discards
|
||||
# the original traceback (see https://bugs.python.org/issue7563),
|
||||
# so sys.exc_info() workaround is used.
|
||||
# This workaround also works in Python 3, but it is not needed,
|
||||
# and it is slower, so in Python 3 we use native `raise`.
|
||||
if six.PY2:
|
||||
exc_info = sys.exc_info()
|
||||
|
||||
self.crawling = False
|
||||
if self.engine is not None:
|
||||
yield self.engine.close()
|
||||
|
||||
if six.PY2:
|
||||
six.reraise(*exc_info)
|
||||
raise
|
||||
|
||||
def _create_spider(self, *args, **kwargs):
|
||||
|
|
@ -119,10 +111,10 @@ class Crawler(object):
|
|||
yield defer.maybeDeferred(self.engine.stop)
|
||||
|
||||
|
||||
class CrawlerRunner(object):
|
||||
class CrawlerRunner:
|
||||
"""
|
||||
This is a convenient helper class that keeps track of, manages and runs
|
||||
crawlers inside an already setup Twisted `reactor`_.
|
||||
crawlers inside an already setup :mod:`~twisted.internet.reactor`.
|
||||
|
||||
The CrawlerRunner object must be instantiated with a
|
||||
:class:`~scrapy.settings.Settings` object.
|
||||
|
|
@ -146,6 +138,7 @@ class CrawlerRunner(object):
|
|||
self._crawlers = set()
|
||||
self._active = set()
|
||||
self.bootstrap_failed = False
|
||||
self._handle_twisted_reactor()
|
||||
|
||||
@property
|
||||
def spiders(self):
|
||||
|
|
@ -216,7 +209,7 @@ class CrawlerRunner(object):
|
|||
return self._create_crawler(crawler_or_spidercls)
|
||||
|
||||
def _create_crawler(self, spidercls):
|
||||
if isinstance(spidercls, six.string_types):
|
||||
if isinstance(spidercls, str):
|
||||
spidercls = self.spider_loader.load(spidercls)
|
||||
return Crawler(spidercls, self.settings)
|
||||
|
||||
|
|
@ -239,18 +232,23 @@ class CrawlerRunner(object):
|
|||
while self._active:
|
||||
yield defer.DeferredList(self._active)
|
||||
|
||||
def _handle_twisted_reactor(self):
|
||||
if self.settings.get("TWISTED_REACTOR"):
|
||||
verify_installed_reactor(self.settings["TWISTED_REACTOR"])
|
||||
|
||||
|
||||
class CrawlerProcess(CrawlerRunner):
|
||||
"""
|
||||
A class to run multiple scrapy crawlers in a process simultaneously.
|
||||
|
||||
This class extends :class:`~scrapy.crawler.CrawlerRunner` by adding support
|
||||
for starting a Twisted `reactor`_ and handling shutdown signals, like the
|
||||
keyboard interrupt command Ctrl-C. It also configures top-level logging.
|
||||
for starting a :mod:`~twisted.internet.reactor` and handling shutdown
|
||||
signals, like the keyboard interrupt command Ctrl-C. It also configures
|
||||
top-level logging.
|
||||
|
||||
This utility should be a better fit than
|
||||
:class:`~scrapy.crawler.CrawlerRunner` if you aren't running another
|
||||
Twisted `reactor`_ within your application.
|
||||
:mod:`~twisted.internet.reactor` within your application.
|
||||
|
||||
The CrawlerProcess object must be instantiated with a
|
||||
:class:`~scrapy.settings.Settings` object.
|
||||
|
|
@ -270,6 +268,7 @@ class CrawlerProcess(CrawlerRunner):
|
|||
log_scrapy_info(self.settings)
|
||||
|
||||
def _signal_shutdown(self, signum, _):
|
||||
from twisted.internet import reactor
|
||||
install_shutdown_handlers(self._signal_kill)
|
||||
signame = signal_names[signum]
|
||||
logger.info("Received %(signame)s, shutting down gracefully. Send again to force ",
|
||||
|
|
@ -277,6 +276,7 @@ class CrawlerProcess(CrawlerRunner):
|
|||
reactor.callFromThread(self._graceful_stop_reactor)
|
||||
|
||||
def _signal_kill(self, signum, _):
|
||||
from twisted.internet import reactor
|
||||
install_shutdown_handlers(signal.SIG_IGN)
|
||||
signame = signal_names[signum]
|
||||
logger.info('Received %(signame)s twice, forcing unclean shutdown',
|
||||
|
|
@ -285,9 +285,9 @@ class CrawlerProcess(CrawlerRunner):
|
|||
|
||||
def start(self, stop_after_crawl=True):
|
||||
"""
|
||||
This method starts a Twisted `reactor`_, adjusts its pool size to
|
||||
:setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache based
|
||||
on :setting:`DNSCACHE_ENABLED` and :setting:`DNSCACHE_SIZE`.
|
||||
This method starts a :mod:`~twisted.internet.reactor`, adjusts its pool
|
||||
size to :setting:`REACTOR_THREADPOOL_MAXSIZE`, and installs a DNS cache
|
||||
based on :setting:`DNSCACHE_ENABLED` and :setting:`DNSCACHE_SIZE`.
|
||||
|
||||
If ``stop_after_crawl`` is True, the reactor will be stopped after all
|
||||
crawlers have finished, using :meth:`join`.
|
||||
|
|
@ -295,6 +295,7 @@ class CrawlerProcess(CrawlerRunner):
|
|||
:param boolean stop_after_crawl: stop or not the reactor when all
|
||||
crawlers have finished
|
||||
"""
|
||||
from twisted.internet import reactor
|
||||
if stop_after_crawl:
|
||||
d = self.join()
|
||||
# Don't start the reactor if the deferreds are already fired
|
||||
|
|
@ -302,34 +303,31 @@ class CrawlerProcess(CrawlerRunner):
|
|||
return
|
||||
d.addBoth(self._stop_reactor)
|
||||
|
||||
reactor.installResolver(self._get_dns_resolver())
|
||||
resolver_class = load_object(self.settings["DNS_RESOLVER"])
|
||||
resolver = create_instance(resolver_class, self.settings, self, reactor=reactor)
|
||||
resolver.install_on_reactor()
|
||||
tp = reactor.getThreadPool()
|
||||
tp.adjustPoolsize(maxthreads=self.settings.getint('REACTOR_THREADPOOL_MAXSIZE'))
|
||||
reactor.addSystemEventTrigger('before', 'shutdown', self.stop)
|
||||
reactor.run(installSignalHandlers=False) # blocking call
|
||||
|
||||
def _get_dns_resolver(self):
|
||||
if self.settings.getbool('DNSCACHE_ENABLED'):
|
||||
cache_size = self.settings.getint('DNSCACHE_SIZE')
|
||||
else:
|
||||
cache_size = 0
|
||||
return CachingThreadedResolver(
|
||||
reactor=reactor,
|
||||
cache_size=cache_size,
|
||||
timeout=self.settings.getfloat('DNS_TIMEOUT')
|
||||
)
|
||||
|
||||
def _graceful_stop_reactor(self):
|
||||
d = self.stop()
|
||||
d.addBoth(self._stop_reactor)
|
||||
return d
|
||||
|
||||
def _stop_reactor(self, _=None):
|
||||
from twisted.internet import reactor
|
||||
try:
|
||||
reactor.stop()
|
||||
except RuntimeError: # raised if already stopped or in shutdown stage
|
||||
pass
|
||||
|
||||
def _handle_twisted_reactor(self):
|
||||
if self.settings.get("TWISTED_REACTOR"):
|
||||
install_reactor(self.settings["TWISTED_REACTOR"])
|
||||
super()._handle_twisted_reactor()
|
||||
|
||||
|
||||
def _get_spider_loader(settings):
|
||||
""" Get SpiderLoader instance from settings """
|
||||
|
|
|
|||
|
|
@ -1,9 +1,7 @@
|
|||
# -*- coding: utf-8 -*-
|
||||
from __future__ import absolute_import
|
||||
import re
|
||||
import logging
|
||||
|
||||
import six
|
||||
from w3lib import html
|
||||
|
||||
from scrapy.exceptions import NotConfigured
|
||||
|
|
@ -49,7 +47,7 @@ class AjaxCrawlMiddleware(object):
|
|||
return response
|
||||
|
||||
# scrapy already handles #! links properly
|
||||
ajax_crawl_request = request.replace(url=request.url+'#!')
|
||||
ajax_crawl_request = request.replace(url=request.url + '#!')
|
||||
logger.debug("Downloading AJAX crawlable %(ajax_crawl_request)s instead of %(request)s",
|
||||
{'ajax_crawl_request': ajax_crawl_request, 'request': request},
|
||||
extra={'spider': spider})
|
||||
|
|
@ -67,7 +65,9 @@ class AjaxCrawlMiddleware(object):
|
|||
|
||||
|
||||
# XXX: move it to w3lib?
|
||||
_ajax_crawlable_re = re.compile(six.u(r'<meta\s+name=["\']fragment["\']\s+content=["\']!["\']/?>'))
|
||||
_ajax_crawlable_re = re.compile(r'<meta\s+name=["\']fragment["\']\s+content=["\']!["\']/?>')
|
||||
|
||||
|
||||
def _has_ajaxcrawlable_meta(text):
|
||||
"""
|
||||
>>> _has_ajaxcrawlable_meta('<html><head><meta name="fragment" content="!"/></head><body></body></html>')
|
||||
|
|
|
|||
|
|
@ -1,12 +1,11 @@
|
|||
import os
|
||||
import six
|
||||
import logging
|
||||
from collections import defaultdict
|
||||
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.http import Response
|
||||
from scrapy.http.cookies import CookieJar
|
||||
from scrapy.utils.python import to_native_str
|
||||
from scrapy.utils.python import to_unicode
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
|
@ -53,7 +52,7 @@ class CookiesMiddleware(object):
|
|||
|
||||
def _debug_cookie(self, request, spider):
|
||||
if self.debug:
|
||||
cl = [to_native_str(c, errors='replace')
|
||||
cl = [to_unicode(c, errors='replace')
|
||||
for c in request.headers.getlist('Cookie')]
|
||||
if cl:
|
||||
cookies = "\n".join("Cookie: {}\n".format(c) for c in cl)
|
||||
|
|
@ -62,7 +61,7 @@ class CookiesMiddleware(object):
|
|||
|
||||
def _debug_set_cookie(self, response, spider):
|
||||
if self.debug:
|
||||
cl = [to_native_str(c, errors='replace')
|
||||
cl = [to_unicode(c, errors='replace')
|
||||
for c in response.headers.getlist('Set-Cookie')]
|
||||
if cl:
|
||||
cookies = "\n".join("Set-Cookie: {}\n".format(c) for c in cl)
|
||||
|
|
@ -82,8 +81,10 @@ class CookiesMiddleware(object):
|
|||
|
||||
def _get_request_cookies(self, jar, request):
|
||||
if isinstance(request.cookies, dict):
|
||||
cookie_list = [{'name': k, 'value': v} for k, v in \
|
||||
six.iteritems(request.cookies)]
|
||||
cookie_list = [
|
||||
{'name': k, 'value': v}
|
||||
for k, v in request.cookies.items()
|
||||
]
|
||||
else:
|
||||
cookie_list = request.cookies
|
||||
|
||||
|
|
|
|||
|
|
@ -4,20 +4,15 @@ and extract the potentially compressed responses that may arrive.
|
|||
|
||||
import bz2
|
||||
import gzip
|
||||
import zipfile
|
||||
import tarfile
|
||||
import logging
|
||||
import tarfile
|
||||
import zipfile
|
||||
from io import BytesIO
|
||||
from tempfile import mktemp
|
||||
|
||||
import six
|
||||
|
||||
try:
|
||||
from cStringIO import StringIO as BytesIO
|
||||
except ImportError:
|
||||
from io import BytesIO
|
||||
|
||||
from scrapy.responsetypes import responsetypes
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
|
|
@ -79,7 +74,7 @@ class DecompressionMiddleware(object):
|
|||
if not response.body:
|
||||
return response
|
||||
|
||||
for fmt, func in six.iteritems(self._formats):
|
||||
for fmt, func in self._formats.items():
|
||||
new_response = func(response)
|
||||
if new_response:
|
||||
logger.debug('Decompressed response with format: %(responsefmt)s',
|
||||
|
|
|
|||
|
|
@ -1,11 +1,19 @@
|
|||
from email.utils import formatdate
|
||||
|
||||
from twisted.internet import defer
|
||||
from twisted.internet.error import TimeoutError, DNSLookupError, \
|
||||
ConnectionRefusedError, ConnectionDone, ConnectError, \
|
||||
ConnectionLost, TCPTimedOutError
|
||||
from twisted.internet.error import (
|
||||
ConnectError,
|
||||
ConnectionDone,
|
||||
ConnectionLost,
|
||||
ConnectionRefusedError,
|
||||
DNSLookupError,
|
||||
TCPTimedOutError,
|
||||
TimeoutError,
|
||||
)
|
||||
from twisted.web.client import ResponseFailed
|
||||
|
||||
from scrapy import signals
|
||||
from scrapy.exceptions import NotConfigured, IgnoreRequest
|
||||
from scrapy.exceptions import IgnoreRequest, NotConfigured
|
||||
from scrapy.utils.misc import load_object
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -26,7 +26,7 @@ class HttpCompressionMiddleware(object):
|
|||
|
||||
def process_request(self, request, spider):
|
||||
request.headers.setdefault('Accept-Encoding',
|
||||
b",".join(ACCEPTED_ENCODINGS))
|
||||
b", ".join(ACCEPTED_ENCODINGS))
|
||||
|
||||
def process_response(self, request, response, spider):
|
||||
|
||||
|
|
@ -37,8 +37,9 @@ class HttpCompressionMiddleware(object):
|
|||
if content_encoding:
|
||||
encoding = content_encoding.pop()
|
||||
decoded_body = self._decode(response.body, encoding.lower())
|
||||
respcls = responsetypes.from_args(headers=response.headers, \
|
||||
url=response.url, body=decoded_body)
|
||||
respcls = responsetypes.from_args(
|
||||
headers=response.headers, url=response.url, body=decoded_body
|
||||
)
|
||||
kwargs = dict(cls=respcls, body=decoded_body)
|
||||
if issubclass(respcls, TextResponse):
|
||||
# force recalculating the encoding until we make sure the
|
||||
|
|
|
|||
|
|
@ -1,10 +1,6 @@
|
|||
import base64
|
||||
from six.moves.urllib.parse import unquote, urlunparse
|
||||
from six.moves.urllib.request import getproxies, proxy_bypass
|
||||
try:
|
||||
from urllib2 import _parse_proxy
|
||||
except ImportError:
|
||||
from urllib.request import _parse_proxy
|
||||
from urllib.parse import unquote, urlunparse
|
||||
from urllib.request import getproxies, proxy_bypass, _parse_proxy
|
||||
|
||||
from scrapy.exceptions import NotConfigured
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
|
|
|
|||
|
|
@ -1,5 +1,5 @@
|
|||
import logging
|
||||
from six.moves.urllib.parse import urljoin
|
||||
from urllib.parse import urljoin, urlparse
|
||||
|
||||
from w3lib.url import safe_url_string
|
||||
|
||||
|
|
@ -7,6 +7,7 @@ from scrapy.http import HtmlResponse
|
|||
from scrapy.utils.response import get_meta_refresh
|
||||
from scrapy.exceptions import IgnoreRequest, NotConfigured
|
||||
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
|
|
@ -70,7 +71,10 @@ class RedirectMiddleware(BaseRedirectMiddleware):
|
|||
if 'Location' not in response.headers or response.status not in allowed_status:
|
||||
return response
|
||||
|
||||
location = safe_url_string(response.headers['location'])
|
||||
location = safe_url_string(response.headers['Location'])
|
||||
if response.headers['Location'].startswith(b'//'):
|
||||
request_scheme = urlparse(request.url).scheme
|
||||
location = request_scheme + '://' + location.lstrip('/')
|
||||
|
||||
redirected_url = urljoin(request.url, location)
|
||||
|
||||
|
|
|
|||
|
|
@ -84,6 +84,6 @@ class RetryMiddleware(object):
|
|||
return retryreq
|
||||
else:
|
||||
stats.inc_value('retry/max_reached')
|
||||
logger.debug("Gave up retrying %(request)s (failed %(retries)d times): %(reason)s",
|
||||
logger.error("Gave up retrying %(request)s (failed %(retries)d times): %(reason)s",
|
||||
{'request': request, 'retries': retries, 'reason': reason},
|
||||
extra={'spider': spider})
|
||||
|
|
|
|||
|
|
@ -5,15 +5,12 @@ enable this middleware and enable the ROBOTSTXT_OBEY setting.
|
|||
"""
|
||||
|
||||
import logging
|
||||
import sys
|
||||
import re
|
||||
|
||||
from twisted.internet.defer import Deferred, maybeDeferred
|
||||
from scrapy.exceptions import NotConfigured, IgnoreRequest
|
||||
from scrapy.http import Request
|
||||
from scrapy.utils.httpobj import urlparse_cached
|
||||
from scrapy.utils.log import failure_to_exc_info
|
||||
from scrapy.utils.python import to_native_str
|
||||
from scrapy.utils.misc import load_object
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
|
|
|||
|
|
@ -1,10 +1,10 @@
|
|||
from __future__ import print_function
|
||||
import os
|
||||
import logging
|
||||
|
||||
from scrapy.utils.job import job_dir
|
||||
from scrapy.utils.request import referer_str, request_fingerprint
|
||||
|
||||
|
||||
class BaseDupeFilter(object):
|
||||
|
||||
@classmethod
|
||||
|
|
@ -49,7 +49,7 @@ class RFPDupeFilter(BaseDupeFilter):
|
|||
return True
|
||||
self.fingerprints.add(fp)
|
||||
if self.file:
|
||||
self.file.write(fp + os.linesep)
|
||||
self.file.write(fp + '\n')
|
||||
|
||||
def request_fingerprint(self, request):
|
||||
return request_fingerprint(request)
|
||||
|
|
|
|||
|
|
@ -7,10 +7,12 @@ new exceptions here without documenting them there.
|
|||
|
||||
# Internal
|
||||
|
||||
|
||||
class NotConfigured(Exception):
|
||||
"""Indicates a missing configuration situation"""
|
||||
pass
|
||||
|
||||
|
||||
class _InvalidOutput(TypeError):
|
||||
"""
|
||||
Indicates an invalid value has been returned by a middleware's processing method.
|
||||
|
|
@ -18,15 +20,19 @@ class _InvalidOutput(TypeError):
|
|||
"""
|
||||
pass
|
||||
|
||||
|
||||
# HTTP and crawling
|
||||
|
||||
|
||||
class IgnoreRequest(Exception):
|
||||
"""Indicates a decision was made not to process a request"""
|
||||
|
||||
|
||||
class DontCloseSpider(Exception):
|
||||
"""Request the spider not to be closed yet"""
|
||||
pass
|
||||
|
||||
|
||||
class CloseSpider(Exception):
|
||||
"""Raise this from callbacks to request the spider to be closed"""
|
||||
|
||||
|
|
@ -34,30 +40,37 @@ class CloseSpider(Exception):
|
|||
super(CloseSpider, self).__init__()
|
||||
self.reason = reason
|
||||
|
||||
|
||||
# Items
|
||||
|
||||
|
||||
class DropItem(Exception):
|
||||
"""Drop item from the item pipeline"""
|
||||
pass
|
||||
|
||||
|
||||
class NotSupported(Exception):
|
||||
"""Indicates a feature or method is not supported"""
|
||||
pass
|
||||
|
||||
|
||||
# Commands
|
||||
|
||||
|
||||
class UsageError(Exception):
|
||||
"""To indicate a command-line usage error"""
|
||||
def __init__(self, *a, **kw):
|
||||
self.print_help = kw.pop('print_help', True)
|
||||
super(UsageError, self).__init__(*a, **kw)
|
||||
|
||||
|
||||
class ScrapyDeprecationWarning(Warning):
|
||||
"""Warning category for deprecated features, since the default
|
||||
DeprecationWarning is silenced on Python 2.7+
|
||||
"""
|
||||
pass
|
||||
|
||||
|
||||
class ContractFail(AssertionError):
|
||||
"""Error raised in case of a failing contract"""
|
||||
pass
|
||||
|
|
|
|||
Some files were not shown because too many files have changed in this diff Show More
Loading…
Reference in New Issue