Merge remote-tracking branch 'scrapy/master' into async-seeds

This commit is contained in:
Adrián Chaves 2025-03-12 02:20:29 +01:00
commit bd73b5437d
68 changed files with 609 additions and 639 deletions

View File

@ -1,7 +1,9 @@
name: Checks
on:
push:
branches: [ master ]
branches:
- master
- '[0-9]+.[0-9]+'
pull_request:
concurrency:

View File

@ -22,7 +22,7 @@ jobs:
- uses: actions/setup-python@v5
with:
python-version: "3.13"
- run: |
- run: |
python -m pip install --upgrade build
python -m build
- name: Publish to PyPI

View File

@ -1,7 +1,9 @@
name: macOS
on:
push:
branches: [ master ]
branches:
- master
- '[0-9]+.[0-9]+'
pull_request:
concurrency:

View File

@ -1,7 +1,9 @@
name: Ubuntu
on:
push:
branches: [ master ]
branches:
- master
- '[0-9]+.[0-9]+'
pull_request:
concurrency:
@ -32,25 +34,25 @@ jobs:
TOXENV: py
- python-version: "3.13"
env:
TOXENV: asyncio
TOXENV: default-reactor
- python-version: pypy3.10
env:
TOXENV: pypy3
# pinned deps
- python-version: 3.9.19
- python-version: "3.9.21"
env:
TOXENV: pinned
- python-version: 3.9.19
- python-version: "3.9.21"
env:
TOXENV: asyncio-pinned
TOXENV: default-reactor-pinned
- python-version: pypy3.10
env:
TOXENV: pypy3-pinned
- python-version: 3.9.19
- python-version: "3.9.21"
env:
TOXENV: extra-deps-pinned
- python-version: 3.9.19
- python-version: "3.9.21"
env:
TOXENV: botocore-pinned
@ -76,7 +78,7 @@ jobs:
if: contains(matrix.python-version, 'pypy') || contains(matrix.env.TOXENV, 'pinned')
run: |
sudo apt-get update
sudo apt-get install libxml2-dev libxslt-dev libjpeg-dev
sudo apt-get install libxml2-dev libxslt-dev
- name: Run tests
env: ${{ matrix.env }}

View File

@ -1,7 +1,9 @@
name: Windows
on:
push:
branches: [ master ]
branches:
- master
- '[0-9]+.[0-9]+'
pull_request:
concurrency:
@ -17,7 +19,7 @@ jobs:
include:
- python-version: "3.9"
env:
TOXENV: windows-pinned
TOXENV: py
- python-version: "3.10"
env:
TOXENV: py
@ -32,7 +34,19 @@ jobs:
TOXENV: py
- python-version: "3.13"
env:
TOXENV: asyncio
TOXENV: default-reactor
# pinned deps
- python-version: "3.9.13"
env:
TOXENV: pinned
- python-version: "3.9.13"
env:
TOXENV: extra-deps-pinned
- python-version: "3.13"
env:
TOXENV: extra-deps
steps:
- uses: actions/checkout@v4

View File

@ -11,3 +11,7 @@ repos:
- id: blacken-docs
additional_dependencies:
- black==24.10.0
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v5.0.0
hooks:
- id: trailing-whitespace

View File

@ -1,6 +1,6 @@
.. image:: https://scrapy.org/img/scrapylogo.png
:target: https://scrapy.org/
======
Scrapy
======

View File

@ -51,7 +51,7 @@ def chdir(tmpdir):
def pytest_addoption(parser):
parser.addoption(
"--reactor",
default="default",
default="asyncio",
choices=["default", "asyncio"],
)
@ -67,17 +67,17 @@ def reactor_pytest(request):
@pytest.fixture(autouse=True)
def only_asyncio(request, reactor_pytest):
if request.node.get_closest_marker("only_asyncio") and reactor_pytest != "asyncio":
pytest.skip("This test is only run with --reactor=asyncio")
if request.node.get_closest_marker("only_asyncio") and reactor_pytest == "default":
pytest.skip("This test is only run without --reactor=default")
@pytest.fixture(autouse=True)
def only_not_asyncio(request, reactor_pytest):
if (
request.node.get_closest_marker("only_not_asyncio")
and reactor_pytest == "asyncio"
and reactor_pytest != "default"
):
pytest.skip("This test is only run without --reactor=asyncio")
pytest.skip("This test is only run with --reactor=default")
@pytest.fixture(autouse=True)
@ -117,7 +117,7 @@ def requires_boto3(request):
def pytest_configure(config):
if config.getoption("--reactor") == "asyncio":
if config.getoption("--reactor") != "default":
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")

View File

@ -16,13 +16,13 @@
</div>
<div class="col-md-4">
<p>
<a href="/login">Login</a>
</p>
</div>
</div>
<div class="row">
<div class="col-md-8">
@ -34,16 +34,16 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
<a class="tag" href="/tag/change/page/1/">change</a>
<a class="tag" href="/tag/deep-thoughts/page/1/">deep-thoughts</a>
<a class="tag" href="/tag/thinking/page/1/">thinking</a>
<a class="tag" href="/tag/world/page/1/">world</a>
</div>
</div>
@ -54,12 +54,12 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
<a class="tag" href="/tag/abilities/page/1/">abilities</a>
<a class="tag" href="/tag/choices/page/1/">choices</a>
</div>
</div>
@ -70,18 +70,18 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
<a class="tag" href="/tag/life/page/1/">life</a>
<a class="tag" href="/tag/live/page/1/">live</a>
<a class="tag" href="/tag/miracle/page/1/">miracle</a>
<a class="tag" href="/tag/miracles/page/1/">miracles</a>
</div>
</div>
@ -92,16 +92,16 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
<a class="tag" href="/tag/aliteracy/page/1/">aliteracy</a>
<a class="tag" href="/tag/books/page/1/">books</a>
<a class="tag" href="/tag/classic/page/1/">classic</a>
<a class="tag" href="/tag/humor/page/1/">humor</a>
</div>
</div>
@ -112,12 +112,12 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
<a class="tag" href="/tag/be-yourself/page/1/">be-yourself</a>
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
</div>
</div>
@ -128,14 +128,14 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
<a class="tag" href="/tag/adulthood/page/1/">adulthood</a>
<a class="tag" href="/tag/success/page/1/">success</a>
<a class="tag" href="/tag/value/page/1/">value</a>
</div>
</div>
@ -146,12 +146,12 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="life,love" / >
<meta class="keywords" itemprop="keywords" content="life,love" / >
<a class="tag" href="/tag/life/page/1/">life</a>
<a class="tag" href="/tag/love/page/1/">love</a>
</div>
</div>
@ -162,16 +162,16 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
<a class="tag" href="/tag/edison/page/1/">edison</a>
<a class="tag" href="/tag/failure/page/1/">failure</a>
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
<a class="tag" href="/tag/paraphrased/page/1/">paraphrased</a>
</div>
</div>
@ -182,10 +182,10 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
<a class="tag" href="/tag/misattributed-eleanor-roosevelt/page/1/">misattributed-eleanor-roosevelt</a>
</div>
</div>
@ -196,73 +196,73 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
<a class="tag" href="/tag/humor/page/1/">humor</a>
<a class="tag" href="/tag/obvious/page/1/">obvious</a>
<a class="tag" href="/tag/simile/page/1/">simile</a>
</div>
</div>
<nav>
<ul class="pager">
<li class="next">
<a href="/page/2/">Next <span aria-hidden="true">&rarr;</span></a>
</li>
</ul>
</nav>
</div>
<div class="col-md-4 tags-box">
<h2>Top Ten tags</h2>
<span class="tag-item">
<a class="tag" style="font-size: 28px" href="/tag/love/">love</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 26px" href="/tag/inspirational/">inspirational</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 26px" href="/tag/life/">life</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 24px" href="/tag/humor/">humor</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 22px" href="/tag/books/">books</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 14px" href="/tag/reading/">reading</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 10px" href="/tag/friendship/">friendship</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 8px" href="/tag/friends/">friends</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 8px" href="/tag/truth/">truth</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 6px" href="/tag/simile/">simile</a>
</span>
</div>
</div>

View File

@ -16,13 +16,13 @@
</div>
<div class="col-md-4">
<p>
<a href="/login">Login</a>
</p>
</div>
</div>
<div class="row">
<div class="col-md-8">
@ -34,16 +34,16 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
<meta class="keywords" itemprop="keywords" content="change,deep-thoughts,thinking,world" / >
<a class="tag" href="/tag/change/page/1/">change</a>
<a class="tag" href="/tag/deep-thoughts/page/1/">deep-thoughts</a>
<a class="tag" href="/tag/thinking/page/1/">thinking</a>
<a class="tag" href="/tag/world/page/1/">world</a>
</div>
</div>
@ -54,12 +54,12 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
<meta class="keywords" itemprop="keywords" content="abilities,choices" / >
<a class="tag" href="/tag/abilities/page/1/">abilities</a>
<a class="tag" href="/tag/choices/page/1/">choices</a>
</div>
</div>
@ -70,18 +70,18 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
<meta class="keywords" itemprop="keywords" content="inspirational,life,live,miracle,miracles" / >
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
<a class="tag" href="/tag/life/page/1/">life</a>
<a class="tag" href="/tag/live/page/1/">live</a>
<a class="tag" href="/tag/miracle/page/1/">miracle</a>
<a class="tag" href="/tag/miracles/page/1/">miracles</a>
</div>
</div>
@ -92,16 +92,16 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
<meta class="keywords" itemprop="keywords" content="aliteracy,books,classic,humor" / >
<a class="tag" href="/tag/aliteracy/page/1/">aliteracy</a>
<a class="tag" href="/tag/books/page/1/">books</a>
<a class="tag" href="/tag/classic/page/1/">classic</a>
<a class="tag" href="/tag/humor/page/1/">humor</a>
</div>
</div>
@ -112,12 +112,12 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
<meta class="keywords" itemprop="keywords" content="be-yourself,inspirational" / >
<a class="tag" href="/tag/be-yourself/page/1/">be-yourself</a>
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
</div>
</div>
@ -128,14 +128,14 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
<meta class="keywords" itemprop="keywords" content="adulthood,success,value" / >
<a class="tag" href="/tag/adulthood/page/1/">adulthood</a>
<a class="tag" href="/tag/success/page/1/">success</a>
<a class="tag" href="/tag/value/page/1/">value</a>
</div>
</div>
@ -146,12 +146,12 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="life,love" / >
<meta class="keywords" itemprop="keywords" content="life,love" / >
<a class="tag" href="/tag/life/page/1/">life</a>
<a class="tag" href="/tag/love/page/1/">love</a>
</div>
</div>
@ -162,16 +162,16 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
<meta class="keywords" itemprop="keywords" content="edison,failure,inspirational,paraphrased" / >
<a class="tag" href="/tag/edison/page/1/">edison</a>
<a class="tag" href="/tag/failure/page/1/">failure</a>
<a class="tag" href="/tag/inspirational/page/1/">inspirational</a>
<a class="tag" href="/tag/paraphrased/page/1/">paraphrased</a>
</div>
</div>
@ -182,10 +182,10 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
<meta class="keywords" itemprop="keywords" content="misattributed-eleanor-roosevelt" / >
<a class="tag" href="/tag/misattributed-eleanor-roosevelt/page/1/">misattributed-eleanor-roosevelt</a>
</div>
</div>
@ -196,73 +196,73 @@
</span>
<div class="tags">
Tags:
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
<meta class="keywords" itemprop="keywords" content="humor,obvious,simile" / >
<a class="tag" href="/tag/humor/page/1/">humor</a>
<a class="tag" href="/tag/obvious/page/1/">obvious</a>
<a class="tag" href="/tag/simile/page/1/">simile</a>
</div>
</div>
<nav>
<ul class="pager">
<li class="next">
<a href="/page/2/">Next <span aria-hidden="true">&rarr;</span></a>
</li>
</ul>
</nav>
</div>
<div class="col-md-4 tags-box">
<h2>Top Ten tags</h2>
<span class="tag-item">
<a class="tag" style="font-size: 28px" href="/tag/love/">love</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 26px" href="/tag/inspirational/">inspirational</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 26px" href="/tag/life/">life</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 24px" href="/tag/humor/">humor</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 22px" href="/tag/books/">books</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 14px" href="/tag/reading/">reading</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 10px" href="/tag/friendship/">friendship</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 8px" href="/tag/friends/">friends</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 8px" href="/tag/truth/">truth</a>
</span>
<span class="tag-item">
<a class="tag" style="font-size: 6px" href="/tag/simile/">simile</a>
</span>
</div>
</div>

View File

@ -361,16 +361,18 @@ method for this purpose. For example:
from copy import deepcopy
from itemadapter import is_item, ItemAdapter
from itemadapter import ItemAdapter
from scrapy import Request
class MultiplyItemsMiddleware:
def process_spider_output(self, response, result, spider):
for item in result:
if is_item(item):
adapter = ItemAdapter(item)
for _ in range(adapter["multiply_by"]):
yield deepcopy(item)
for item_or_request in result:
if isinstance(item_or_request, Request):
continue
adapter = ItemAdapter(item)
for _ in range(adapter["multiply_by"]):
yield deepcopy(item)
Does Scrapy support IPv6 addresses?
-----------------------------------
@ -410,7 +412,7 @@ How can I make a blank request?
-------------------------------
.. code-block:: python
from scrapy import Request

View File

@ -111,7 +111,7 @@ Once you've installed `Anaconda`_ or `Miniconda`_, install Scrapy with::
To install Scrapy on Windows using ``pip``:
.. warning::
This installation method requires “Microsoft Visual C++” for installing some
This installation method requires “Microsoft Visual C++” for installing some
Scrapy dependencies, which demands significantly more disk space than Anaconda.
#. Download and execute `Microsoft C++ Build Tools`_ to install the Visual Studio Installer.
@ -123,7 +123,7 @@ To install Scrapy on Windows using ``pip``:
#. Check the installation details and make sure following packages are selected as optional components:
* **MSVC** (e.g MSVC v142 - VS 2019 C++ x64/x86 build tools (v14.23) )
* **Windows SDK** (e.g Windows 10 SDK (10.0.18362.0))
#. Install the Visual Studio Build Tools.

View File

@ -32,7 +32,8 @@ This is an example where two add-ons are enabled in a project's
Writing your own add-ons
========================
Add-ons are Python classes that include one or both of the following methods:
Add-ons are :ref:`components <topics-components>` that include one or both of
the following methods:
.. method:: update_settings(settings)
@ -54,20 +55,6 @@ Add-ons are Python classes that include one or both of the following methods:
:param settings: The settings object storing Scrapy/component configuration
:type settings: :class:`~scrapy.settings.BaseSettings`
They can also have the following method:
.. classmethod:: from_crawler(cls, crawler)
:noindex:
If present, this class method is called to create an add-on instance
from a :class:`~scrapy.crawler.Crawler`. It must return a new instance
of the add-on. The crawler object provides access to all Scrapy core
components like settings and signals; it is a way for the add-on to access
them and hook its functionality into Scrapy.
:param crawler: The crawler that uses this add-on
:type crawler: :class:`~scrapy.crawler.Crawler`
The settings set by the add-on should use the ``addon`` priority (see
:ref:`populating-settings` and :func:`scrapy.settings.BaseSettings.set`)::

View File

@ -12,10 +12,11 @@ extensions and middlewares.
Crawler API
===========
The main entry point to Scrapy API is the :class:`~scrapy.crawler.Crawler`
object, passed to extensions through the ``from_crawler`` class method. This
object provides access to all Scrapy core components, and it's the only way for
extensions to access them and hook their functionality into Scrapy.
The main entry point to the Scrapy API is the :class:`~scrapy.crawler.Crawler`
object, which :ref:`components <topics-components>` can :ref:`get for
initialization <from-crawler>`. It provides access to all Scrapy core
components, and it is the only way for components to access them and hook their
functionality into Scrapy.
.. module:: scrapy.crawler
:synopsis: The Scrapy crawler
@ -88,7 +89,7 @@ how you :ref:`configure the downloader middlewares
The execution engine, which coordinates the core crawling logic
between the scheduler, downloader and spiders.
Some extension may want to access the Scrapy engine, to inspect or
Some extension may want to access the Scrapy engine, to inspect or
modify the downloader and scheduler behaviour, although this is an
advanced use and this API is not yet stable.

View File

@ -16,15 +16,19 @@ asyncio reactor <install-asyncio>`, you may use :mod:`asyncio` and
Installing the asyncio reactor
==============================
To enable :mod:`asyncio` support, set the :setting:`TWISTED_REACTOR` setting to
``'twisted.internet.asyncioreactor.AsyncioSelectorReactor'``.
To enable :mod:`asyncio` support, your :setting:`TWISTED_REACTOR` setting needs
to be set to ``'twisted.internet.asyncioreactor.AsyncioSelectorReactor'``,
which is the default value.
If you are using :class:`~scrapy.crawler.CrawlerRunner`, you also need to
install the :class:`~twisted.internet.asyncioreactor.AsyncioSelectorReactor`
reactor manually. You can do that using
:func:`~scrapy.utils.reactor.install_reactor`::
:func:`~scrapy.utils.reactor.install_reactor`:
install_reactor('twisted.internet.asyncioreactor.AsyncioSelectorReactor')
.. skip: next
.. code-block:: python
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
.. _asyncio-preinstalled-reactor:
@ -144,3 +148,14 @@ Using custom asyncio loops
You can also use custom asyncio event loops with the asyncio reactor. Set the
:setting:`ASYNCIO_EVENT_LOOP` setting to the import path of the desired event
loop class to use it instead of the default asyncio event loop.
.. _disable-asyncio:
Switching to a non-asyncio reactor
==================================
If for some reason your code doesn't work with the asyncio reactor, you can use
a different reactor by setting the :setting:`TWISTED_REACTOR` setting to its
import path (e.g. ``'twisted.internet.epollreactor.EPollReactor'``) or to
``None``, which will use the default reactor for your platform.

View File

@ -9,6 +9,8 @@ A Scrapy component is any class whose objects are built using
That includes the classes that you may assign to the following settings:
- :setting:`ADDONS`
- :setting:`DNS_RESOLVER`
- :setting:`DOWNLOAD_HANDLERS`
@ -41,10 +43,80 @@ Third-party Scrapy components may also let you define additional Scrapy
components, usually configurable through :ref:`settings <topics-settings>`, to
modify their behavior.
.. _from-crawler:
Initializing from the crawler
=============================
Any Scrapy component may optionally define the following class method:
.. classmethod:: from_crawler(cls, crawler: scrapy.crawler.Crawler, *args, **kwargs)
Return an instance of the component based on *crawler*.
*args* and *kwargs* are component-specific arguments that some components
receive. However, most components do not get any arguments, and instead
:ref:`use settings <component-settings>`.
If a component class defines this method, this class method is called to
create any instance of the component.
The *crawler* object provides access to all Scrapy core components like
:ref:`settings <topics-settings>` and :ref:`signals <topics-signals>`,
allowing the component to access them and hook its functionality into
Scrapy.
.. _component-settings:
Settings
========
Components can be configured through :ref:`settings <topics-settings>`.
Components can read any setting from the
:attr:`~scrapy.crawler.Crawler.settings` attribute of the
:class:`~scrapy.crawler.Crawler` object they can :ref:`get for initialization
<from-crawler>`. That includes both built-in and custom settings.
For example:
.. code-block:: python
class MyExtension:
@classmethod
def from_crawler(cls, crawler):
settings = crawler.settings
return cls(settings.getbool("LOG_ENABLED"))
def __init__(self, log_is_enabled=False):
if log_is_enabled:
print("log is enabled!")
Components do not need to declare their custom settings programmatically.
However, they should document them, so that users know they exist and how to
use them.
It is a good practice to prefix custom settings with the name of the component,
to avoid collisions with custom settings of other existing (or future)
components. For example, an extension called ``WarcCaching`` could prefix its
custom settings with ``WARC_CACHING_``.
Another good practice, mainly for components meant for :ref:`component priority
dictionaries <component-priority-dictionaries>`, is to provide a boolean setting
called ``<PREFIX>_ENABLED`` (e.g. ``WARC_CACHING_ENABLED``) to allow toggling
that component on and off without changing the component priority dictionary
setting. You can usually check the value of such a setting during
initialization, and if ``False``, raise
:exc:`~scrapy.exceptions.NotConfigured`.
When choosing a name for a custom setting, it is also a good idea to have a
look at the names of :ref:`built-in settings <topics-settings-ref>`, to try to
maintain consistency with them.
.. _enforce-component-requirements:
Enforcing component requirements
================================
Enforcing requirements
======================
Sometimes, your components may only be intended to work under certain
conditions. For example, they may require a minimum version of Scrapy to work as
@ -58,8 +130,8 @@ In the case of :ref:`downloader middlewares <topics-downloader-middleware>`,
:ref:`extensions <topics-extensions>`, :ref:`item pipelines
<topics-item-pipeline>`, and :ref:`spider middlewares
<topics-spider-middleware>`, you should raise
:exc:`scrapy.exceptions.NotConfigured`, passing a description of the issue as a
parameter to the exception so that it is printed in the logs, for the user to
:exc:`~scrapy.exceptions.NotConfigured`, passing a description of the issue as
a parameter to the exception so that it is printed in the logs, for the user to
see. For other components, feel free to raise whatever other exception feels
right to you; for example, :exc:`RuntimeError` would make sense for a Scrapy
version mismatch, while :exc:`ValueError` may be better if the issue is the

View File

@ -61,12 +61,8 @@ particular setting. See each middleware documentation for more info.
Writing your own downloader middleware
======================================
Each downloader middleware is a Python class that defines one or more of the
methods defined below.
The main entry point is the ``from_crawler`` class method, which receives a
:class:`~scrapy.crawler.Crawler` instance. The :class:`~scrapy.crawler.Crawler`
object gives you access, for example, to the :ref:`settings <topics-settings>`.
Each downloader middleware is a :ref:`component <topics-components>` that
defines one or more of these methods:
.. module:: scrapy.downloadermiddlewares
@ -167,17 +163,6 @@ object gives you access, for example, to the :ref:`settings <topics-settings>`.
:param spider: the spider for which this request is intended
:type spider: :class:`~scrapy.Spider` object
.. method:: from_crawler(cls, crawler)
If present, this classmethod is called to create a middleware instance
from a :class:`~scrapy.crawler.Crawler`. It must return a new instance
of the middleware. Crawler object provides access to all Scrapy core
components like settings and signals; it is a way for middleware to
access them and hook its functionality into Scrapy.
:param crawler: crawler that uses this middleware
:type crawler: :class:`~scrapy.crawler.Crawler` object
.. _topics-downloader-middleware-ref:
Built-in downloader middleware reference

View File

@ -50,9 +50,9 @@ And here is how to use it to send an e-mail (without attachments):
MailSender class reference
==========================
MailSender is the preferred class to use for sending emails from Scrapy, as it
uses :doc:`Twisted non-blocking IO <twisted:core/howto/defer-intro>`, like the
rest of the framework.
The MailSender :ref:`components <topics-components>` is the preferred class to
use for sending emails from Scrapy, as it uses :doc:`Twisted non-blocking IO
<twisted:core/howto/defer-intro>`, like the rest of the framework.
.. class:: MailSender(smtphost=None, mailfrom=None, smtpuser=None, smtppass=None, smtpport=None)
@ -81,14 +81,6 @@ rest of the framework.
:param smtpssl: enforce using a secure SSL connection
:type smtpssl: bool
.. classmethod:: from_crawler(crawler)
Instantiate using a :class:`scrapy.Crawler` instance, which will
respect :ref:`these Scrapy settings <topics-email-settings>`.
:param crawler: the crawler
:type settings: :class:`scrapy.Crawler` object
.. method:: send(to, subject, body, cc=None, attachs=(), mimetype='text/plain', charset=None)
Send email to the given recipients.

View File

@ -224,7 +224,7 @@ BaseItemExporter
.. [1] Not all exporters respect the specified field order.
.. [2] When using :ref:`item objects <item-types>` that do not expose
all their possible fields, exporters that do not support exporting
a different subset of fields per item will only export the fields
a different subset of fields per item will only export the fields
found in the first item exported.
.. attribute:: export_empty_fields

View File

@ -4,34 +4,21 @@
Extensions
==========
The extensions framework provides a mechanism for inserting your own
custom functionality into Scrapy.
Extensions are :ref:`components <topics-components>` that allow inserting your
own custom functionality into Scrapy.
Extensions are just regular classes.
Unlike other components, extensions do not have a specific role in Scrapy. They
are “wildcard” components that can be used for anything that does not fit the
role of any other type of component.
Extension settings
==================
Loading and activating extensions
=================================
Extensions use the :ref:`Scrapy settings <topics-settings>` to manage their
settings, just like any other Scrapy code.
Extensions are loaded at startup by creating a single instance of the extension
class per spider being run.
It is customary for extensions to prefix their settings with their own name, to
avoid collision with existing (and future) extensions. For example, a
hypothetical extension to handle `Google Sitemaps`_ would use settings like
``GOOGLESITEMAP_ENABLED``, ``GOOGLESITEMAP_DEPTH``, and so on.
.. _Google Sitemaps: https://en.wikipedia.org/wiki/Sitemaps
Loading & activating extensions
===============================
Extensions are loaded and activated at startup by instantiating a single
instance of the extension class per spider being run. All the extension
initialization code must be performed in the class ``__init__`` method.
To make an extension available, add it to the :setting:`EXTENSIONS` setting in
your Scrapy settings. In :setting:`EXTENSIONS`, each extension is represented
by a string: the full Python path to the extension's class name. For example:
To enable an extension, add it to the :setting:`EXTENSIONS` setting. For
example:
.. code-block:: python
@ -40,55 +27,24 @@ by a string: the full Python path to the extension's class name. For example:
"scrapy.extensions.telnet.TelnetConsole": 500,
}
As you can see, the :setting:`EXTENSIONS` setting is a dict where the keys are
the extension paths, and their values are the orders, which define the
extension *loading* order. The :setting:`EXTENSIONS` setting is merged with the
:setting:`EXTENSIONS_BASE` setting defined in Scrapy (and not meant to be
overridden) and then sorted by order to get the final sorted list of enabled
extensions.
:setting:`EXTENSIONS` is merged with :setting:`EXTENSIONS_BASE` (not meant to
be overridden), and the priorities in the resulting value determine the
*loading* order.
As extensions typically do not depend on each other, their loading order is
irrelevant in most cases. This is why the :setting:`EXTENSIONS_BASE` setting
defines all extensions with the same order (``0``). However, this feature can
be exploited if you need to add an extension which depends on other extensions
already loaded.
Available, enabled and disabled extensions
==========================================
Not all available extensions will be enabled. Some of them usually depend on a
particular setting. For example, the HTTP Cache extension is available by default
but disabled unless the :setting:`HTTPCACHE_ENABLED` setting is set.
Disabling an extension
======================
In order to disable an extension that comes enabled by default (i.e. those
included in the :setting:`EXTENSIONS_BASE` setting) you must set its order to
``None``. For example:
.. code-block:: python
EXTENSIONS = {
"scrapy.extensions.corestats.CoreStats": None,
}
defines all extensions with the same order (``0``). However, you may need to
carefully use priorities if you add an extension that depends on other
extensions being already loaded.
Writing your own extension
==========================
Each extension is a Python class. The main entry point for a Scrapy extension
(this also includes middlewares and pipelines) is the ``from_crawler``
class method which receives a ``Crawler`` instance. Through the Crawler object
you can access settings, signals, stats, and also control the crawling behaviour.
Each extension is a :ref:`component <topics-components>`.
Typically, extensions connect to :ref:`signals <topics-signals>` and perform
tasks triggered by them.
Finally, if the ``from_crawler`` method raises the
:exc:`~scrapy.exceptions.NotConfigured` exception, the extension will be
disabled. Otherwise, the extension will be enabled.
Sample extension
----------------
@ -256,14 +212,14 @@ Spider state extension
Manages spider state data by loading it before a crawl and saving it after.
Give a value to the :setting:`JOBDIR` setting to enable this extension.
When enabled, this extension manages the :attr:`~scrapy.Spider.state`
When enabled, this extension manages the :attr:`~scrapy.Spider.state`
attribute of your :class:`~scrapy.Spider` instance:
- When your spider closes (:signal:`spider_closed`), the contents of its
:attr:`~scrapy.Spider.state` attribute are serialized into a file named
- When your spider closes (:signal:`spider_closed`), the contents of its
:attr:`~scrapy.Spider.state` attribute are serialized into a file named
``spider.state`` in the :setting:`JOBDIR` folder.
- When your spider opens (:signal:`spider_opened`), if a previously-generated
``spider.state`` file exists in the :setting:`JOBDIR` folder, it is loaded
- When your spider opens (:signal:`spider_opened`), if a previously-generated
``spider.state`` file exists in the :setting:`JOBDIR` folder, it is loaded
into the :attr:`~scrapy.Spider.state` attribute.
@ -291,8 +247,8 @@ settings:
.. note::
When a certain closing condition is met, requests which are
currently in the downloader queue (up to :setting:`CONCURRENT_REQUESTS`
When a certain closing condition is met, requests which are
currently in the downloader queue (up to :setting:`CONCURRENT_REQUESTS`
requests) are still processed.
.. setting:: CLOSESPIDER_TIMEOUT

View File

@ -180,7 +180,7 @@ FTP supports two different connection modes: `active or passive
mode by default. To use the active connection mode instead, set the
:setting:`FEED_STORAGE_FTP_ACTIVE` setting to ``True``.
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
storage backend is: ``True``.
.. caution:: The value ``True`` in ``overwrite`` will cause you to lose the
@ -222,7 +222,7 @@ feeds using these settings:
- :setting:`AWS_ENDPOINT_URL`
- :setting:`AWS_REGION_NAME`
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
storage backend is: ``True``.
.. caution:: The value ``True`` in ``overwrite`` will cause you to lose the
@ -255,7 +255,7 @@ You can set a *Project ID* and *Access Control List (ACL)* through the following
- :setting:`FEED_STORAGE_GCS_ACL`
- :setting:`GCS_PROJECT_ID`
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
The default value for the ``overwrite`` key in the :setting:`FEEDS` for this
storage backend is: ``True``.
.. caution:: The value ``True`` in ``overwrite`` will cause you to lose the
@ -587,8 +587,8 @@ FEED_STORE_EMPTY
Default: ``True``
Whether to export empty feeds (i.e. feeds with no items).
If ``False``, and there are no items to export, no new files are created and
existing files are not modified, even if the :ref:`overwrite feed option
If ``False``, and there are no items to export, no new files are created and
existing files are not modified, even if the :ref:`overwrite feed option
<feed-options>` is enabled.
.. setting:: FEED_STORAGES

View File

@ -23,7 +23,8 @@ Typical uses of item pipelines are:
Writing your own item pipeline
==============================
Each item pipeline component is a Python class that must implement the following method:
Each item pipeline is a :ref:`component <topics-components>` that must
implement the following method:
.. method:: process_item(self, item, spider)
@ -60,17 +61,6 @@ Additionally, they may also implement the following methods:
:param spider: the spider which was closed
:type spider: :class:`~scrapy.Spider` object
.. classmethod:: from_crawler(cls, crawler)
If present, this class method is called to create a pipeline instance
from a :class:`~scrapy.crawler.Crawler`. It must return a new instance
of the pipeline. Crawler object provides access to all Scrapy core
components like settings and signals; it is a way for pipeline to
access them and hook its functionality into Scrapy.
:param crawler: crawler that uses this pipeline
:type crawler: :class:`~scrapy.crawler.Crawler` object
Item pipeline example
=====================
@ -139,8 +129,8 @@ In this example we'll write items to MongoDB_ using pymongo_.
MongoDB address and database name are specified in Scrapy settings;
MongoDB collection is named after item class.
The main point of this example is to show how to use :meth:`from_crawler`
method and how to clean up the resources properly.
The main point of this example is to show how to :ref:`get the crawler
<from-crawler>` and how to clean up the resources properly.
.. skip: next
.. code-block:: python

View File

@ -384,9 +384,8 @@ Supporting All Item Types
In code that receives an item, such as methods of :ref:`item pipelines
<topics-item-pipeline>` or :ref:`spider middlewares
<topics-spider-middleware>`, it is a good practice to use the
:class:`~itemadapter.ItemAdapter` class and the
:func:`~itemadapter.is_item` function to write code that works for
any supported item type.
:class:`~itemadapter.ItemAdapter` class to write code that works for any
supported item type.
Other classes related to items
==============================

View File

@ -266,9 +266,9 @@ e.g. in the spider's ``__init__`` method:
If you run this spider again then INFO messages from
``scrapy.spidermiddlewares.httperror`` logger will be gone.
You can also filter log records by :class:`~logging.LogRecord` data. For
You can also filter log records by :class:`~logging.LogRecord` data. For
example, you can filter log records by message content using a substring or
a regular expression. Create a :class:`logging.Filter` subclass
a regular expression. Create a :class:`logging.Filter` subclass
and equip it with a regular expression pattern to
filter out unwanted messages:
@ -284,8 +284,8 @@ filter out unwanted messages:
if match:
return False
A project-level filter may be attached to the root
handler created by Scrapy, this is a wieldy way to
A project-level filter may be attached to the root
handler created by Scrapy, this is a wieldy way to
filter all loggers in different parts of the project
(middlewares, spider, etc.):
@ -301,7 +301,7 @@ filter all loggers in different parts of the project
for handler in logging.root.handlers:
handler.addFilter(ContentFilter())
Alternatively, you may choose a specific logger
Alternatively, you may choose a specific logger
and hide it without affecting other loggers:
.. code-block:: python

View File

@ -70,7 +70,7 @@ The advantage of using the :class:`ImagesPipeline` for image files is that you
can configure some extra functions like generating thumbnails and filtering
the images based on their size.
The Images Pipeline requires Pillow_ 7.1.0 or greater. It is used for
The Images Pipeline requires Pillow_ 8.0.0 or greater. It is used for
thumbnailing and normalizing images to JPEG/RGB format.
.. _Pillow: https://github.com/python-pillow/Pillow
@ -414,7 +414,7 @@ class name. E.g. given pipeline class called MyPipeline you can set setting key:
and pipeline class MyPipeline will have expiration time set to 180.
The last modified time from the file is used to determine the age of the file in days,
The last modified time from the file is used to determine the age of the file in days,
which is then compared to the set expiration time to determine if the file is expired.
.. _topics-images-thumbnails:
@ -519,7 +519,7 @@ See here the methods that you can override in your custom Files Pipeline:
In addition to ``response``, this method receives the original
:class:`request <scrapy.Request>`,
:class:`info <scrapy.pipelines.media.MediaPipeline.SpiderInfo>` and
:class:`info <scrapy.pipelines.media.MediaPipeline.SpiderInfo>` and
:class:`item <scrapy.Item>`
You can override this method to customize the download path of each file.
@ -541,9 +541,9 @@ See here the methods that you can override in your custom Files Pipeline:
def file_path(self, request, response=None, info=None, *, item=None):
return "files/" + PurePosixPath(urlparse_cached(request).path).name
Similarly, you can use the ``item`` to determine the file path based on some item
Similarly, you can use the ``item`` to determine the file path based on some item
property.
By default the :meth:`file_path` method returns
``full/<request URL hash>.<extension>``.
@ -677,7 +677,7 @@ See here the methods that you can override in your custom Images Pipeline:
In addition to ``response``, this method receives the original
:class:`request <scrapy.Request>`,
:class:`info <scrapy.pipelines.media.MediaPipeline.SpiderInfo>` and
:class:`info <scrapy.pipelines.media.MediaPipeline.SpiderInfo>` and
:class:`item <scrapy.Item>`
You can override this method to customize the download path of each file.
@ -699,9 +699,9 @@ See here the methods that you can override in your custom Images Pipeline:
def file_path(self, request, response=None, info=None, *, item=None):
return "files/" + PurePosixPath(urlparse_cached(request).path).name
Similarly, you can use the ``item`` to determine the file path based on some item
Similarly, you can use the ``item`` to determine the file path based on some item
property.
By default the :meth:`file_path` method returns
``full/<request URL hash>.<extension>``.

View File

@ -309,7 +309,7 @@ Here are some tips to keep in mind when dealing with these kinds of sites:
services like `ProxyMesh`_. An open source alternative is `scrapoxy`_, a
super proxy that you can attach your own proxies to.
* use a ban avoidance service, such as `Zyte API`_, which provides a `Scrapy
plugin <https://github.com/scrapy-plugins/scrapy-zyte-api>`__ and additional
plugin <https://github.com/scrapy-plugins/scrapy-zyte-api>`__ and additional
features, like `AI web scraping <https://www.zyte.com/ai-web-scraping/>`__
If you are still unable to prevent your bot getting banned, consider contacting

View File

@ -463,35 +463,17 @@ import path.
Writing your own request fingerprinter
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
A request fingerprinter is a class that must implement the following method:
A request fingerprinter is a :ref:`component <topics-components>` that must
implement the following method:
.. currentmodule:: None
.. method:: fingerprint(self, request)
.. method:: fingerprint(self, request: scrapy.Request)
Return a :class:`bytes` object that uniquely identifies *request*.
See also :ref:`request-fingerprint-restrictions`.
:param request: request to fingerprint
:type request: scrapy.Request
Additionally, it may also implement the following method:
.. classmethod:: from_crawler(cls, crawler)
:noindex:
If present, this class method is called to create a request fingerprinter
instance from a :class:`~scrapy.crawler.Crawler` object. It must return a
new instance of the request fingerprinter.
*crawler* provides access to all Scrapy core components like settings and
signals; it is a way for the request fingerprinter to access them and hook
its functionality into Scrapy.
:param crawler: crawler that uses this request fingerprinter
:type crawler: :class:`~scrapy.crawler.Crawler` object
.. currentmodule:: scrapy.http
The :meth:`fingerprint` method of the default request fingerprinter,

View File

@ -559,7 +559,7 @@ For example, suppose you want to extract all ``<p>`` elements inside ``<div>``
elements. First, you would get all ``<div>`` elements:
.. code-block:: pycon
>>> divs = response.xpath("//div")
At first, you may be tempted to use the following approach, which is wrong, as
@ -610,7 +610,7 @@ As it turns out, Scrapy selectors allow you to chain selectors, so most of the t
you can just select by class using CSS and then switch to XPath when needed:
.. code-block:: pycon
>>> from scrapy import Selector
>>> sel = Selector(
... text='<div class="hero shout"><time datetime="2014-07-23 19:00">Special date</time></div>'
@ -1032,7 +1032,7 @@ whereas the CSS lookup is translated into XPath and thus runs more efficiently,
so performance-wise its uses are limited to situations that are not easily
described with CSS selectors.
Parsel also simplifies adding your own XPath extensions with
Parsel also simplifies adding your own XPath extensions with
:func:`~parsel.xpathfuncs.set_xpathfunc`.
.. _topics-selectors-ref:

View File

@ -204,7 +204,7 @@ How to access settings
.. highlight:: python
In a spider, the settings are available through ``self.settings``:
In a spider, settings are available through ``self.settings``:
.. code-block:: python
@ -217,37 +217,17 @@ In a spider, the settings are available through ``self.settings``:
.. note::
The ``settings`` attribute is set in the base Spider class after the spider
is initialized. If you want to use the settings before the initialization
is initialized. If you want to use settings before the initialization
(e.g., in your spider's ``__init__()`` method), you'll need to override the
:meth:`~scrapy.Spider.from_crawler` method.
Settings can be accessed through the :attr:`scrapy.crawler.Crawler.settings`
attribute of the Crawler that is passed to ``from_crawler`` method in
extensions, middlewares and item pipelines:
:ref:`Components <topics-components>` can also :ref:`access settings
<component-settings>`.
.. code-block:: python
class MyExtension:
def __init__(self, log_is_enabled=False):
if log_is_enabled:
print("log is enabled!")
@classmethod
def from_crawler(cls, crawler):
settings = crawler.settings
return cls(settings.getbool("LOG_ENABLED"))
The settings object can be used like a dict (e.g.,
``settings['LOG_ENABLED']``), but it's usually preferred to extract the setting
in the format you need it to avoid type errors, using one of the methods
provided by the :class:`~scrapy.settings.Settings` API.
Rationale for setting names
===========================
Setting names are usually prefixed with the component that they configure. For
example, proper setting names for a fictional robots.txt extension would be
``ROBOTSTXT_ENABLED``, ``ROBOTSTXT_OBEY``, ``ROBOTSTXT_CACHEDIR``, etc.
The ``settings`` object can be used like a :class:`dict` (e.g.
``settings["LOG_ENABLED"]``). However, to support non-string setting values,
which may be passed from the command line as strings, it is recommended to use
one of the methods provided by the :class:`~scrapy.settings.Settings` API.
.. _component-priority-dictionaries:
@ -1211,7 +1191,8 @@ EXTENSIONS
Default:: ``{}``
A dict containing the extensions enabled in your project, and their orders.
:ref:`Component priority dictionary <component-priority-dictionaries>` of
enabled extensions. See :ref:`topics-extensions`.
.. setting:: EXTENSIONS_BASE
@ -1998,7 +1979,7 @@ TWISTED_REACTOR
.. versionadded:: 2.0
Default: ``None``
Default: ``"twisted.internet.asyncioreactor.AsyncioSelectorReactor"``
Import path of a given :mod:`~twisted.internet.reactor`.
@ -2083,17 +2064,19 @@ which raises :exc:`Exception`, becomes:
self.crawler.engine.close_spider(self, "timeout")
The default value of the :setting:`TWISTED_REACTOR` setting is ``None``, which
means that Scrapy will use the existing reactor if one is already installed, or
install the default reactor defined by Twisted for the current platform. This
is to maintain backward compatibility and avoid possible problems caused by
using a non-default reactor.
If this setting is set ``None``, Scrapy will use the existing reactor if one is
already installed, or install the default reactor defined by Twisted for the
current platform.
.. versionchanged:: 2.7
The :command:`startproject` command now sets this setting to
``twisted.internet.asyncioreactor.AsyncioSelectorReactor`` in the generated
``settings.py`` file.
.. versionchanged:: VERSION
The default value was changed from ``None`` to
``"twisted.internet.asyncioreactor.AsyncioSelectorReactor"``.
For additional information, see :doc:`core/howto/choosing-reactor`.

View File

@ -63,28 +63,13 @@ particular setting. See each middleware documentation for more info.
Writing your own spider middleware
==================================
Each spider middleware is a Python class that defines one or more of the
methods defined below.
The main entry point is the ``from_crawler`` class method, which receives a
:class:`~scrapy.crawler.Crawler` instance. The :class:`~scrapy.crawler.Crawler`
object gives you access, for example, to the :ref:`settings <topics-settings>`.
Each spider middleware is a :ref:`component <topics-components>` that defines
one or more of these methods:
.. module:: scrapy.spidermiddlewares
.. class:: SpiderMiddleware
.. method:: from_crawler(cls, crawler)
If present, this classmethod is called to create a middleware instance
from a :class:`~scrapy.crawler.Crawler`. It must return a new instance
of the middleware. Crawler object provides access to all Scrapy core
components like settings and signals; it is a way for middleware to
access them and hook its functionality into Scrapy.
:param crawler: crawler that uses this middleware
:type crawler: :class:`~scrapy.crawler.Crawler` object
.. method:: process_seeds(seeds: AsyncIterator[Any], /) -> AsyncIterator[Any]
:async:

View File

@ -86,7 +86,7 @@ Available Stats Collectors
Besides the basic :class:`StatsCollector` there are other Stats Collectors
available in Scrapy which extend the basic Stats Collector. You can select
which Stats Collector to use through the :setting:`STATS_CLASS` setting. The
default Stats Collector used is the :class:`MemoryStatsCollector`.
default Stats Collector used is the :class:`MemoryStatsCollector`.
.. currentmodule:: scrapy.statscollectors

View File

@ -374,10 +374,6 @@ ignore = [
"B904",
# Use capitalized environment variable
"SIM112",
# Temporarily silenced PT rules
# Use a regular `assert` instead of unittest-style `assertEqual`
"PT009",
]
[tool.ruff.lint.per-file-ignores]

View File

@ -6,7 +6,7 @@ import json
import logging
from typing import TYPE_CHECKING, Any, TypeVar, overload
from itemadapter import ItemAdapter, is_item
from itemadapter import ItemAdapter
from twisted.internet.defer import Deferred, maybeDeferred
from w3lib.url import is_url
@ -211,10 +211,10 @@ class Command(BaseRunSpiderCommand):
) -> tuple[list[Any], list[Request], argparse.Namespace, int, Spider, CallbackT]:
items, requests = [], []
for x in spider_output:
if is_item(x):
items.append(x)
elif isinstance(x, Request):
if isinstance(x, Request):
requests.append(x)
else:
items.append(x)
return items, requests, opts, depth, spider, callback
def run_callback(

View File

@ -8,7 +8,6 @@ from collections import deque
from collections.abc import AsyncIterable, Iterator
from typing import TYPE_CHECKING, Any, TypeVar, Union, cast
from itemadapter import is_item
from twisted.internet.defer import Deferred, inlineCallbacks
from twisted.python.failure import Failure
@ -298,17 +297,10 @@ class Scraper:
if isinstance(output, Request):
assert self.crawler.engine is not None # typing
self.crawler.engine.crawl(request=output)
elif is_item(output):
return self.start_itemproc(output, response=response)
elif output is None:
pass
else:
typename = type(output).__name__
logger.error(
"Spider must return request, item, or None, got %(typename)r in %(request)s",
{"request": request, "typename": typename},
extra={"spider": spider},
)
return self.start_itemproc(output, response=response)
return None
def start_itemproc(self, item: Any, *, response: Response | None) -> Deferred[Any]:

View File

@ -356,12 +356,12 @@ class PythonItemExporter(BaseItemExporter):
def _serialize_value(self, value: Any) -> Any:
if isinstance(value, Item):
return self.export_item(value)
if isinstance(value, (str, bytes)):
return to_unicode(value, encoding=self.encoding)
if is_item(value):
return dict(self._serialize_item(value))
if is_listlike(value):
return [self._serialize_value(v) for v in value]
if isinstance(value, (str, bytes)):
return to_unicode(value, encoding=self.encoding)
return value
def _serialize_item(self, item: Any) -> Iterable[tuple[str | bytes, Any]]:

View File

@ -68,7 +68,7 @@ class ImagesPipeline(FilesPipeline):
self._Image = Image
except ImportError:
raise NotConfigured(
"ImagesPipeline requires installing Pillow 4.0.0 or later"
"ImagesPipeline requires installing Pillow 8.0.0 or later"
)
super().__init__(

View File

@ -343,7 +343,7 @@ TELNETCONSOLE_HOST = "127.0.0.1"
TELNETCONSOLE_USERNAME = "scrapy"
TELNETCONSOLE_PASSWORD = None
TWISTED_REACTOR = None
TWISTED_REACTOR = "twisted.internet.asyncioreactor.AsyncioSelectorReactor"
SPIDER_CONTRACTS = {}
SPIDER_CONTRACTS_BASE = {

View File

@ -6,7 +6,7 @@
from scrapy import signals
# useful for handling different item types with a single interface
from itemadapter import is_item, ItemAdapter
from itemadapter import ItemAdapter
class ${ProjectName}SpiderMiddleware:

View File

@ -90,5 +90,4 @@ ROBOTSTXT_OBEY = True
#HTTPCACHE_STORAGE = "scrapy.extensions.httpcache.FilesystemCacheStorage"
# Set settings whose default value is deprecated to a future-proof value
TWISTED_REACTOR = "twisted.internet.asyncioreactor.AsyncioSelectorReactor"
FEED_EXPORT_ENCODING = "utf-8"

View File

@ -182,11 +182,9 @@ def log_scrapy_info(settings: Settings) -> None:
def log_reactor_info() -> None:
from twisted.internet import reactor
from twisted.internet import asyncioreactor, reactor
logger.debug("Using reactor: %s.%s", reactor.__module__, reactor.__class__.__name__)
from twisted.internet import asyncioreactor
if isinstance(reactor, asyncioreactor.AsyncioSelectorReactor):
logger.debug(
"Using asyncio event loop: %s.%s",

View File

@ -28,12 +28,12 @@ class ScrapyJSONEncoder(json.JSONEncoder):
return str(o)
if isinstance(o, defer.Deferred):
return str(o)
if is_item(o):
return ItemAdapter(o).asdict()
if isinstance(o, Request):
return f"<{type(o).__name__} {o.method} {o.url}>"
if isinstance(o, Response):
return f"<{type(o).__name__} {o.status} {o.url}>"
if is_item(o):
return ItemAdapter(o).asdict()
return super().default(o)

View File

@ -18,6 +18,7 @@ from twisted.trial.unittest import SkipTest
from scrapy.exceptions import ScrapyDeprecationWarning
from scrapy.utils.boto import is_botocore_available
from scrapy.utils.deprecate import create_deprecated_class
from scrapy.utils.reactor import is_asyncio_reactor_installed
from scrapy.utils.spider import DefaultSpider
if TYPE_CHECKING:
@ -109,6 +110,19 @@ def get_ftp_content_and_delete(
TestSpider = create_deprecated_class("TestSpider", DefaultSpider)
def get_reactor_settings() -> dict[str, Any]:
"""Return a settings dict that works with the installed reactor.
``Crawler._apply_settings()`` checks that the installed reactor matches the
settings, so tests that run the crawler in the current process may need to
pass a correct ``"TWISTED_REACTOR"`` setting value when creating it.
"""
settings: dict[str, Any] = {}
if not is_asyncio_reactor_installed():
settings["TWISTED_REACTOR"] = None
return settings
def get_crawler(
spidercls: type[Spider] | None = None,
settings_dict: dict[str, Any] | None = None,
@ -120,9 +134,12 @@ def get_crawler(
"""
from scrapy.crawler import CrawlerRunner
# Set by default settings that prevent deprecation warnings.
settings: dict[str, Any] = {}
settings.update(settings_dict or {})
# When needed, useful settings can be added here, e.g. ones that prevent
# deprecation warnings.
settings: dict[str, Any] = {
**get_reactor_settings(),
**(settings_dict or {}),
}
runner = CrawlerRunner(settings)
crawler = runner.create_crawler(spidercls or DefaultSpider)
crawler._apply_settings()
@ -156,7 +173,7 @@ def assert_samelines(
category=ScrapyDeprecationWarning,
stacklevel=2,
)
testcase.assertEqual(text1.splitlines(), text2.splitlines(), msg)
testcase.assertEqual(text1.splitlines(), text2.splitlines(), msg) # noqa: PT009
def get_from_asyncio_queue(value: _T) -> Awaitable[_T]:

View File

@ -11,7 +11,7 @@ SEP-004: Library API
====================
.. note:: the library API has been implemented, but slightly different from
proposed in this SEP. You can run a Scrapy crawler inside a Twisted
reactor, but not outside it.
reactor, but not outside it.
Introduction
============

View File

@ -96,7 +96,7 @@ specified, else utf-8 is used) and returns a new unicode object. E.g:
``clean_spaces``
----------------
Converts multispaces into single spaces for the given string. E.g:
::

View File

@ -73,8 +73,8 @@ Alternative Public API Proposal
- ``ItemLoader.get_stored_values()`` or ``ItemLoader.get_values()`` *(returns the ``ItemLoader values)*
- ``ItemLoader.get_output_value()``
- ``ItemLoader.get_input_processor()`` or ``ItemLoader.get_in_processor()`` *(short version)*
- ``ItemLoader.get_output_processor()`` or ``ItemLoader.get_out_processor()`` *(short version)*
- ``ItemLoader.get_input_processor()`` or ``ItemLoader.get_in_processor()`` *(short version)*
- ``ItemLoader.get_output_processor()`` or ``ItemLoader.get_out_processor()`` *(short version)*
- ``ItemLoader.context``

View File

@ -21,7 +21,7 @@ Current flaws and inconsistencies
2. Link extractors are inflexible and hard to maintain, link
processing/filtering is tightly coupled. (e.g. canonicalize)
3. Isn't possible to crawl an url directly from command line because the Spider
does not know which callback use.
does not know which callback use.
These flaws will be corrected by the changes proposed in this SEP.
@ -55,7 +55,7 @@ Request Extractors
Request Extractors takes response object and determines which requests follow.
This is an enhancement to ``LinkExtractors`` which returns urls (links),
Request Extractors return Request objects.
Request Extractors return Request objects.
Request Processors
------------------

View File

@ -200,7 +200,7 @@ the same spider:
# extract item from response
return item
The Spider Middleware that implements spider code
The Spider Middleware that implements spider code
=================================================
There's gonna be one middleware that will take care of calling the proper
@ -625,7 +625,7 @@ Resolved:
not the original one (think of redirections), but it does carry the ``meta``
of the original one. The original one may not be available anymore (in
memory) if we're using a persistent scheduler., but in that case it would be
the deserialized request from the persistent scheduler queue.
the deserialized request from the persistent scheduler queue.
- No - this would make implementation more complex and we're not sure it's
really needed

View File

@ -1,14 +1,8 @@
import asyncio
import sys
import scrapy
from scrapy.crawler import CrawlerProcess
from scrapy.utils.reactor import install_reactor
from twisted.internet import asyncioreactor
if sys.platform == "win32":
asyncio.set_event_loop_policy(asyncio.WindowsSelectorEventLoopPolicy())
asyncioreactor.install(asyncio.get_event_loop())
import scrapy # noqa: E402
from scrapy.crawler import CrawlerProcess # noqa: E402
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
class NoRequestsSpider(scrapy.Spider):

View File

@ -1,4 +1,5 @@
from twisted.internet import reactor # noqa: F401
from twisted.python import log
import scrapy
from scrapy.crawler import CrawlerProcess
@ -13,5 +14,6 @@ class NoRequestsSpider(scrapy.Spider):
process = CrawlerProcess(settings={})
process.crawl(NoRequestsSpider)
d = process.crawl(NoRequestsSpider)
d.addErrback(log.err)
process.start()

View File

@ -1,4 +1,5 @@
from twisted.internet import selectreactor
from twisted.python import log
import scrapy
from scrapy.crawler import CrawlerProcess
@ -15,5 +16,6 @@ class NoRequestsSpider(scrapy.Spider):
process = CrawlerProcess(settings={})
process.crawl(NoRequestsSpider)
d = process.crawl(NoRequestsSpider)
d.addErrback(log.err)
process.start()

View File

@ -1,3 +1,9 @@
# ruff: noqa: E402
from scrapy.utils.reactor import install_reactor
install_reactor("twisted.internet.asyncioreactor.AsyncioSelectorReactor")
from urllib.parse import urlparse
from twisted.internet import reactor

View File

@ -9,7 +9,7 @@ from scrapy import Spider
from scrapy.crawler import Crawler, CrawlerRunner
from scrapy.exceptions import NotConfigured
from scrapy.settings import BaseSettings, Settings
from scrapy.utils.test import get_crawler
from scrapy.utils.test import get_crawler, get_reactor_settings
class SimpleAddon:
@ -105,6 +105,7 @@ class TestAddonManager(unittest.TestCase):
}
settings_dict = {
"ADDONS": {get_addon_cls(config): 1},
**get_reactor_settings(),
}
crawler = get_crawler(settings_dict=settings_dict)
assert crawler.settings.getint("KEY") == 15
@ -119,6 +120,7 @@ class TestAddonManager(unittest.TestCase):
settings_dict = {
"KEY": 20, # priority=project
"ADDONS": {get_addon_cls(config): 1},
**get_reactor_settings(),
}
settings = Settings(settings_dict)
settings.set("KEY", 0, priority="default")
@ -196,6 +198,7 @@ class TestAddonManager(unittest.TestCase):
return spider
settings = Settings()
settings.setdict(get_reactor_settings())
settings.set("KEY", "default", priority="default")
runner = CrawlerRunner(settings)
crawler = runner.create_crawler(MySpider)

View File

@ -18,7 +18,7 @@ from scrapy.exceptions import StopDownload
from scrapy.http import Request
from scrapy.http.response import Response
from scrapy.utils.python import to_unicode
from scrapy.utils.test import get_crawler
from scrapy.utils.test import get_crawler, get_reactor_settings
from tests import NON_EXISTING_RESOLVABLE
from tests.mockserver import MockServer
from tests.spiders import (
@ -411,7 +411,7 @@ with multiples lines
@defer.inlineCallbacks
def test_crawl_multiple(self):
runner = CrawlerRunner()
runner = CrawlerRunner(get_reactor_settings())
runner.crawl(
SimpleSpider,
self.mockserver.url("/status?n=200"),

View File

@ -25,7 +25,7 @@ from scrapy.settings import Settings, default_settings
from scrapy.spiderloader import SpiderLoader
from scrapy.utils.log import configure_logging, get_scrapy_root_handler
from scrapy.utils.spider import DefaultSpider
from scrapy.utils.test import get_crawler
from scrapy.utils.test import get_crawler, get_reactor_settings
from tests.mockserver import MockServer, get_mockserver_env
BASE_SETTINGS: dict[str, Any] = {}
@ -35,6 +35,7 @@ def get_raw_crawler(spidercls=None, settings_dict=None):
"""get_crawler alternative that only calls the __init__ method of the
crawler."""
settings = Settings()
settings.setdict(get_reactor_settings())
settings.setdict(settings_dict or {})
return Crawler(spidercls or DefaultSpider, settings)
@ -48,7 +49,12 @@ class TestBaseCrawler(unittest.TestCase):
class TestCrawler(TestBaseCrawler):
def test_populate_spidercls_settings(self):
spider_settings = {"TEST1": "spider", "TEST2": "spider"}
project_settings = {**BASE_SETTINGS, "TEST1": "project", "TEST3": "project"}
project_settings = {
**BASE_SETTINGS,
"TEST1": "project",
"TEST3": "project",
**get_reactor_settings(),
}
class CustomSettingsSpider(DefaultSpider):
custom_settings = spider_settings
@ -581,7 +587,7 @@ class NoRequestsSpider(scrapy.Spider):
@pytest.mark.usefixtures("reactor_pytest")
class TestCrawlerRunnerHasSpider(unittest.TestCase):
def _runner(self):
return CrawlerRunner()
return CrawlerRunner(get_reactor_settings())
@inlineCallbacks
def test_crawler_runner_bootstrap_successful(self):
@ -626,13 +632,7 @@ class TestCrawlerRunnerHasSpider(unittest.TestCase):
@inlineCallbacks
def test_crawler_runner_asyncio_enabled_true(self):
if self.reactor_pytest == "asyncio":
CrawlerRunner(
settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
}
)
else:
if self.reactor_pytest == "default":
runner = CrawlerRunner(
settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
@ -643,6 +643,12 @@ class TestCrawlerRunnerHasSpider(unittest.TestCase):
match=r"The installed reactor \(.*?\) does not match the requested one \(.*?\)",
):
yield runner.crawl(NoRequestsSpider)
else:
CrawlerRunner(
settings={
"TWISTED_REACTOR": "twisted.internet.asyncioreactor.AsyncioSelectorReactor",
}
)
class ScriptRunnerMixin:
@ -672,7 +678,7 @@ class TestCrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
assert "Spider closed (finished)" in log
assert (
"Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor"
not in log
in log
)
def test_multi(self):
@ -680,18 +686,17 @@ class TestCrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
assert "Spider closed (finished)" in log
assert (
"Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor"
not in log
in log
)
assert "ReactorAlreadyInstalledError" not in log
def test_reactor_default(self):
log = self.run_script("reactor_default.py")
assert "Spider closed (finished)" in log
assert "Spider closed (finished)" not in log
assert (
"Using reactor: twisted.internet.asyncioreactor.AsyncioSelectorReactor"
not in log
)
assert "ReactorAlreadyInstalledError" not in log
"does not match the requested one "
"(twisted.internet.asyncioreactor.AsyncioSelectorReactor)"
) in log
def test_reactor_default_twisted_reactor_select(self):
log = self.run_script("reactor_default_twisted_reactor_select.py")
@ -716,8 +721,11 @@ class TestCrawlerProcessSubprocess(ScriptRunnerMixin, unittest.TestCase):
def test_reactor_select(self):
log = self.run_script("reactor_select.py")
assert "Spider closed (finished)" in log
assert "ReactorAlreadyInstalledError" not in log
assert "Spider closed (finished)" not in log
assert (
"does not match the requested one "
"(twisted.internet.asyncioreactor.AsyncioSelectorReactor)"
) in log
def test_reactor_select_twisted_reactor_select(self):
log = self.run_script("reactor_select_twisted_reactor_select.py")

View File

@ -33,7 +33,7 @@ class TestScrapyUtils:
tox_config_file_path = Path(__file__).parent / ".." / "tox.ini"
config_parser = ConfigParser()
config_parser.read(tox_config_file_path)
pattern = r"Twisted\[http2\]==([\d.]+)"
pattern = r"Twisted==([\d.]+)"
match = re.search(pattern, config_parser["pinned"]["deps"])
pinned_twisted_version_string = match[1]

View File

@ -307,7 +307,7 @@ class TestHttp(unittest.TestCase, ABC):
@defer.inlineCallbacks
def test_timeout_download_from_spider_nodata_rcvd(self):
if self.reactor_pytest == "asyncio" and sys.platform == "win32":
if self.reactor_pytest != "default" and sys.platform == "win32":
# https://twistedmatrix.com/trac/ticket/10279
raise unittest.SkipTest(
"This test produces DirtyReactorAggregateError on Windows with asyncio"
@ -322,7 +322,7 @@ class TestHttp(unittest.TestCase, ABC):
@defer.inlineCallbacks
def test_timeout_download_from_spider_server_hangs(self):
if self.reactor_pytest == "asyncio" and sys.platform == "win32":
if self.reactor_pytest != "default" and sys.platform == "win32":
# https://twistedmatrix.com/trac/ticket/10279
raise unittest.SkipTest(
"This test produces DirtyReactorAggregateError on Windows with asyncio"
@ -1136,7 +1136,7 @@ class TestFTPBase(unittest.TestCase):
class TestFTP(TestFTPBase):
def test_invalid_credentials(self):
if self.reactor_pytest == "asyncio" and sys.platform == "win32":
if self.reactor_pytest != "default" and sys.platform == "win32":
raise unittest.SkipTest(
"This test produces DirtyReactorAggregateError on Windows with asyncio"
)

View File

@ -1,3 +1,4 @@
from typing import Any
from unittest import mock
import pytest
@ -171,7 +172,11 @@ Disallow: /some/randome/page.html
middleware = RobotsTxtMiddleware(self.crawler)
middleware._logerror = mock.MagicMock(side_effect=middleware._logerror)
deferred = middleware.process_request(Request("http://site.local"), None)
deferred.addCallback(lambda _: self.assertTrue(middleware._logerror.called))
def check_called(_: Any) -> None:
assert middleware._logerror.called
deferred.addCallback(check_called)
return deferred
def test_robotstxt_immediate_error(self):
@ -202,7 +207,11 @@ Disallow: /some/randome/page.html
mw_module_logger.error = mock.MagicMock()
d = self.assertNotIgnored(Request("http://site.local/allowed"), middleware)
d.addCallback(lambda _: self.assertFalse(mw_module_logger.error.called))
def check_not_called(_: Any) -> None:
assert not mw_module_logger.error.called # type: ignore[attr-defined]
d.addCallback(check_not_called)
return d
def test_robotstxt_user_agent_setting(self):

View File

@ -64,7 +64,7 @@ class CrawlTestCase(TestCase):
@defer.inlineCallbacks
def test_delay(self):
crawler = CrawlerRunner().create_crawler(DownloaderSlotsSettingsTestSpider)
crawler = get_crawler(DownloaderSlotsSettingsTestSpider)
yield crawler.crawl(mockserver=self.mockserver)
slots = crawler.engine.downloader.slots
times = crawler.spider.times

View File

@ -434,10 +434,12 @@ class TestEngine(TestEngineBase):
e = ExecutionEngine(get_crawler(MySpider), lambda _: None)
yield e.open_spider(MySpider(), [])
e.start()
def cb(exc: BaseException) -> None:
assert str(exc), "Engine already running"
try:
yield self.assertFailure(e.start(), RuntimeError).addBoth(
lambda exc: self.assertEqual(str(exc), "Engine already running")
)
yield self.assertFailure(e.start(), RuntimeError).addBoth(cb)
finally:
yield e.stop()

View File

@ -1,9 +1,11 @@
import datetime
import typing
import unittest
from __future__ import annotations
import datetime
import unittest
from typing import Any, Callable
from scrapy.crawler import Crawler
from scrapy.extensions.periodic_log import PeriodicLog
from scrapy.utils.test import get_crawler
from .spiders import MetaSpider
@ -59,9 +61,8 @@ class CustomPeriodicLog(PeriodicLog):
self.stats._stats = stats_dump_2
def extension(settings=None):
crawler = Crawler(MetaSpider, settings=settings)
crawler._apply_settings()
def extension(settings: dict[str, Any] | None = None) -> CustomPeriodicLog:
crawler = get_crawler(MetaSpider, settings)
return CustomPeriodicLog.from_crawler(crawler)
@ -94,7 +95,7 @@ class TestPeriodicLog(unittest.TestCase):
ext.spider_closed(spider, reason="finished")
return ext, a, b
def check(settings: dict, condition: typing.Callable):
def check(settings: dict[str, Any], condition: Callable) -> None:
ext, a, b = emulate(settings)
assert list(a["delta"].keys()) == [
k for k, v in ext.stats._stats.items() if condition(k, v)
@ -151,7 +152,7 @@ class TestPeriodicLog(unittest.TestCase):
ext.spider_closed(spider, reason="finished")
return ext, a, b
def check(settings: dict, condition: typing.Callable):
def check(settings: dict[str, Any], condition: Callable) -> None:
ext, a, b = emulate(settings)
assert list(a["stats"].keys()) == [
k for k, v in ext.stats._stats.items() if condition(k, v)

View File

@ -3,18 +3,22 @@ from __future__ import annotations
import shutil
from pathlib import Path
from tempfile import mkdtemp
from typing import TYPE_CHECKING, Any
from testfixtures import LogCapture
from twisted.internet import defer
from twisted.trial.unittest import TestCase
from w3lib.url import add_or_replace_parameter
from scrapy import signals
from scrapy.crawler import CrawlerRunner
from scrapy import Spider, signals
from scrapy.utils.misc import load_object
from scrapy.utils.test import get_crawler
from tests.mockserver import MockServer
from tests.spiders import SimpleSpider
if TYPE_CHECKING:
from scrapy.crawler import Crawler
class MediaDownloadSpider(SimpleSpider):
name = "mediadownload"
@ -80,7 +84,6 @@ class TestFileDownloadCrawl(TestCase):
"ITEM_PIPELINES": {self.pipeline_class: 1},
self.store_setting_key: str(self.tmpmediastore),
}
self.runner = CrawlerRunner(self.settings)
self.items = []
def tearDown(self):
@ -90,10 +93,12 @@ class TestFileDownloadCrawl(TestCase):
def _on_item_scraped(self, item):
self.items.append(item)
def _create_crawler(self, spider_class, runner=None, **kwargs):
if runner is None:
runner = self.runner
crawler = runner.create_crawler(spider_class, **kwargs)
def _create_crawler(
self, spider_class: type[Spider], settings: dict[str, Any] | None = None
) -> Crawler:
if settings is None:
settings = self.settings
crawler = get_crawler(spider_class, settings)
crawler.signals.connect(self._on_item_scraped, signals.item_scraped)
return crawler
@ -175,10 +180,11 @@ class TestFileDownloadCrawl(TestCase):
@defer.inlineCallbacks
def test_download_media_redirected_allowed(self):
settings = dict(self.settings)
settings.update({"MEDIA_ALLOW_REDIRECTS": True})
runner = CrawlerRunner(settings)
crawler = self._create_crawler(RedirectedMediaDownloadSpider, runner=runner)
settings = {
**self.settings,
"MEDIA_ALLOW_REDIRECTS": True,
}
crawler = self._create_crawler(RedirectedMediaDownloadSpider, settings)
with LogCapture() as log:
yield crawler.crawl(
self.mockserver.url("/files/images/"),
@ -201,8 +207,7 @@ class TestFileDownloadCrawl(TestCase):
**self.settings,
"ITEM_PIPELINES": {ExceptionRaisingMediaPipeline: 1},
}
runner = CrawlerRunner(settings)
crawler = self._create_crawler(MediaDownloadSpider, runner=runner)
crawler = self._create_crawler(MediaDownloadSpider, settings)
with LogCapture() as log:
yield crawler.crawl(
self.mockserver.url("/files/images/"),

View File

@ -266,17 +266,11 @@ class TestFilesPipeline(unittest.TestCase):
class FilesPipelineTestCaseFieldsMixin:
def setup_method(self):
self.tempdir = mkdtemp()
def teardown_method(self):
rmtree(self.tempdir)
def test_item_fields_default(self):
def test_item_fields_default(self, tmp_path):
url = "http://www.example.com/files/1.txt"
item = self.item_class(name="item1", file_urls=[url])
pipeline = FilesPipeline.from_crawler(
get_crawler(None, {"FILES_STORE": self.tempdir})
get_crawler(None, {"FILES_STORE": tmp_path})
)
requests = list(pipeline.get_media_requests(item, None))
assert requests[0].url == url
@ -286,14 +280,14 @@ class FilesPipelineTestCaseFieldsMixin:
assert files == [results[0][1]]
assert isinstance(item, self.item_class)
def test_item_fields_override_settings(self):
def test_item_fields_override_settings(self, tmp_path):
url = "http://www.example.com/files/1.txt"
item = self.item_class(name="item1", custom_file_urls=[url])
pipeline = FilesPipeline.from_crawler(
get_crawler(
None,
{
"FILES_STORE": self.tempdir,
"FILES_STORE": tmp_path,
"FILES_URLS_FIELD": "custom_file_urls",
"FILES_RESULT_FIELD": "custom_files",
},
@ -368,13 +362,7 @@ class TestFilesPipelineCustomSettings:
("FILES_RESULT_FIELD", "FILES_RESULT_FIELD", "files_result_field"),
}
def setup_method(self):
self.tempdir = mkdtemp()
def teardown_method(self):
rmtree(self.tempdir)
def _generate_fake_settings(self, prefix=None):
def _generate_fake_settings(self, tmp_path, prefix=None):
def random_string():
return "".join([chr(random.randint(97, 123)) for _ in range(10)])
@ -382,7 +370,7 @@ class TestFilesPipelineCustomSettings:
"FILES_EXPIRES": random.randint(100, 1000),
"FILES_URLS_FIELD": random_string(),
"FILES_RESULT_FIELD": random_string(),
"FILES_STORE": self.tempdir,
"FILES_STORE": tmp_path,
}
if not prefix:
return settings
@ -400,16 +388,16 @@ class TestFilesPipelineCustomSettings:
return UserDefinedFilePipeline
def test_different_settings_for_different_instances(self):
def test_different_settings_for_different_instances(self, tmp_path):
"""
If there are different instances with different settings they should keep
different settings.
"""
custom_settings = self._generate_fake_settings()
custom_settings = self._generate_fake_settings(tmp_path)
another_pipeline = FilesPipeline.from_crawler(
get_crawler(None, custom_settings)
)
one_pipeline = FilesPipeline(self.tempdir, crawler=get_crawler(None))
one_pipeline = FilesPipeline(tmp_path, crawler=get_crawler(None))
for pipe_attr, settings_attr, pipe_ins_attr in self.file_cls_attr_settings_map:
default_value = self.default_cls_settings[pipe_attr]
assert getattr(one_pipeline, pipe_attr) == default_value
@ -417,24 +405,24 @@ class TestFilesPipelineCustomSettings:
assert default_value != custom_value
assert getattr(another_pipeline, pipe_ins_attr) == custom_value
def test_subclass_attributes_preserved_if_no_settings(self):
def test_subclass_attributes_preserved_if_no_settings(self, tmp_path):
"""
If subclasses override class attributes and there are no special settings those values should be kept.
"""
pipe_cls = self._generate_fake_pipeline()
pipe = pipe_cls.from_crawler(get_crawler(None, {"FILES_STORE": self.tempdir}))
pipe = pipe_cls.from_crawler(get_crawler(None, {"FILES_STORE": tmp_path}))
for pipe_attr, settings_attr, pipe_ins_attr in self.file_cls_attr_settings_map:
custom_value = getattr(pipe, pipe_ins_attr)
assert custom_value != self.default_cls_settings[pipe_attr]
assert getattr(pipe, pipe_ins_attr) == getattr(pipe, pipe_attr)
def test_subclass_attrs_preserved_custom_settings(self):
def test_subclass_attrs_preserved_custom_settings(self, tmp_path):
"""
If file settings are defined but they are not defined for subclass
settings should be preserved.
"""
pipeline_cls = self._generate_fake_pipeline()
settings = self._generate_fake_settings()
settings = self._generate_fake_settings(tmp_path)
pipeline = pipeline_cls.from_crawler(get_crawler(None, settings))
for pipe_attr, settings_attr, pipe_ins_attr in self.file_cls_attr_settings_map:
value = getattr(pipeline, pipe_ins_attr)
@ -442,7 +430,7 @@ class TestFilesPipelineCustomSettings:
assert value != self.default_cls_settings[pipe_attr]
assert value == setting_value
def test_no_custom_settings_for_subclasses(self):
def test_no_custom_settings_for_subclasses(self, tmp_path):
"""
If there are no settings for subclass and no subclass attributes, pipeline should use
attributes of base class.
@ -452,14 +440,14 @@ class TestFilesPipelineCustomSettings:
pass
user_pipeline = UserDefinedFilesPipeline.from_crawler(
get_crawler(None, {"FILES_STORE": self.tempdir})
get_crawler(None, {"FILES_STORE": tmp_path})
)
for pipe_attr, settings_attr, pipe_ins_attr in self.file_cls_attr_settings_map:
# Values from settings for custom pipeline should be set on pipeline instance.
custom_value = self.default_cls_settings.get(pipe_attr.upper())
assert getattr(user_pipeline, pipe_ins_attr) == custom_value
def test_custom_settings_for_subclasses(self):
def test_custom_settings_for_subclasses(self, tmp_path):
"""
If there are custom settings for subclass and NO class attributes, pipeline should use custom
settings.
@ -469,7 +457,7 @@ class TestFilesPipelineCustomSettings:
pass
prefix = UserDefinedFilesPipeline.__name__.upper()
settings = self._generate_fake_settings(prefix=prefix)
settings = self._generate_fake_settings(tmp_path, prefix=prefix)
user_pipeline = UserDefinedFilesPipeline.from_crawler(
get_crawler(None, settings)
)
@ -479,14 +467,14 @@ class TestFilesPipelineCustomSettings:
assert custom_value != self.default_cls_settings[pipe_attr]
assert getattr(user_pipeline, pipe_inst_attr) == custom_value
def test_custom_settings_and_class_attrs_for_subclasses(self):
def test_custom_settings_and_class_attrs_for_subclasses(self, tmp_path):
"""
If there are custom settings for subclass AND class attributes
setting keys are preferred and override attributes.
"""
pipeline_cls = self._generate_fake_pipeline()
prefix = pipeline_cls.__name__.upper()
settings = self._generate_fake_settings(prefix=prefix)
settings = self._generate_fake_settings(tmp_path, prefix=prefix)
user_pipeline = pipeline_cls.from_crawler(get_crawler(None, settings))
for (
pipe_cls_attr,
@ -497,13 +485,13 @@ class TestFilesPipelineCustomSettings:
assert custom_value != self.default_cls_settings[pipe_cls_attr]
assert getattr(user_pipeline, pipe_inst_attr) == custom_value
def test_cls_attrs_with_DEFAULT_prefix(self):
def test_cls_attrs_with_DEFAULT_prefix(self, tmp_path):
class UserDefinedFilesPipeline(FilesPipeline):
DEFAULT_FILES_RESULT_FIELD = "this"
DEFAULT_FILES_URLS_FIELD = "that"
pipeline = UserDefinedFilesPipeline.from_crawler(
get_crawler(None, {"FILES_STORE": self.tempdir})
get_crawler(None, {"FILES_STORE": tmp_path})
)
assert (
pipeline.files_result_field
@ -514,12 +502,12 @@ class TestFilesPipelineCustomSettings:
== UserDefinedFilesPipeline.DEFAULT_FILES_URLS_FIELD
)
def test_user_defined_subclass_default_key_names(self):
def test_user_defined_subclass_default_key_names(self, tmp_path):
"""Test situation when user defines subclass of FilesPipeline,
but uses attribute names for default pipeline (without prefixing
them with pipeline class name).
"""
settings = self._generate_fake_settings()
settings = self._generate_fake_settings(tmp_path)
class UserPipe(FilesPipeline):
pass

View File

@ -314,13 +314,7 @@ class TestImagesPipelineCustomSettings:
"IMAGES_RESULT_FIELD": "images",
}
def setup_method(self):
self.tempdir = mkdtemp()
def teardown_method(self):
rmtree(self.tempdir)
def _generate_fake_settings(self, prefix=None):
def _generate_fake_settings(self, tmp_path, prefix=None):
"""
:param prefix: string for setting keys
:return: dictionary of image pipeline settings
@ -331,7 +325,7 @@ class TestImagesPipelineCustomSettings:
settings = {
"IMAGES_EXPIRES": random.randint(100, 1000),
"IMAGES_STORE": self.tempdir,
"IMAGES_STORE": tmp_path,
"IMAGES_RESULT_FIELD": random_string(),
"IMAGES_URLS_FIELD": random_string(),
"IMAGES_MIN_WIDTH": random.randint(1, 1000),
@ -368,13 +362,13 @@ class TestImagesPipelineCustomSettings:
return UserDefinedImagePipeline
def test_different_settings_for_different_instances(self):
def test_different_settings_for_different_instances(self, tmp_path):
"""
If there are two instances of ImagesPipeline class with different settings, they should
have different settings.
"""
custom_settings = self._generate_fake_settings()
default_sts_pipe = ImagesPipeline(self.tempdir, crawler=get_crawler(None))
custom_settings = self._generate_fake_settings(tmp_path)
default_sts_pipe = ImagesPipeline(tmp_path, crawler=get_crawler(None))
user_sts_pipe = ImagesPipeline.from_crawler(get_crawler(None, custom_settings))
for pipe_attr, settings_attr in self.img_cls_attribute_names:
expected_default_value = self.default_pipeline_settings.get(pipe_attr)
@ -385,14 +379,14 @@ class TestImagesPipelineCustomSettings:
)
assert getattr(user_sts_pipe, pipe_attr.lower()) == custom_value
def test_subclass_attrs_preserved_default_settings(self):
def test_subclass_attrs_preserved_default_settings(self, tmp_path):
"""
If image settings are not defined at all subclass of ImagePipeline takes values
from class attributes.
"""
pipeline_cls = self._generate_fake_pipeline_subclass()
pipeline = pipeline_cls.from_crawler(
get_crawler(None, {"IMAGES_STORE": self.tempdir})
get_crawler(None, {"IMAGES_STORE": tmp_path})
)
for pipe_attr, settings_attr in self.img_cls_attribute_names:
# Instance attribute (lowercase) must be equal to class attribute (uppercase).
@ -400,13 +394,13 @@ class TestImagesPipelineCustomSettings:
assert attr_value != self.default_pipeline_settings[pipe_attr]
assert attr_value == getattr(pipeline, pipe_attr)
def test_subclass_attrs_preserved_custom_settings(self):
def test_subclass_attrs_preserved_custom_settings(self, tmp_path):
"""
If image settings are defined but they are not defined for subclass default
values taken from settings should be preserved.
"""
pipeline_cls = self._generate_fake_pipeline_subclass()
settings = self._generate_fake_settings()
settings = self._generate_fake_settings(tmp_path)
pipeline = pipeline_cls.from_crawler(get_crawler(None, settings))
for pipe_attr, settings_attr in self.img_cls_attribute_names:
# Instance attribute (lowercase) must be equal to
@ -416,7 +410,7 @@ class TestImagesPipelineCustomSettings:
setings_value = settings.get(settings_attr)
assert value == setings_value
def test_no_custom_settings_for_subclasses(self):
def test_no_custom_settings_for_subclasses(self, tmp_path):
"""
If there are no settings for subclass and no subclass attributes, pipeline should use
attributes of base class.
@ -426,14 +420,14 @@ class TestImagesPipelineCustomSettings:
pass
user_pipeline = UserDefinedImagePipeline.from_crawler(
get_crawler(None, {"IMAGES_STORE": self.tempdir})
get_crawler(None, {"IMAGES_STORE": tmp_path})
)
for pipe_attr, settings_attr in self.img_cls_attribute_names:
# Values from settings for custom pipeline should be set on pipeline instance.
custom_value = self.default_pipeline_settings.get(pipe_attr.upper())
assert getattr(user_pipeline, pipe_attr.lower()) == custom_value
def test_custom_settings_for_subclasses(self):
def test_custom_settings_for_subclasses(self, tmp_path):
"""
If there are custom settings for subclass and NO class attributes, pipeline should use custom
settings.
@ -443,7 +437,7 @@ class TestImagesPipelineCustomSettings:
pass
prefix = UserDefinedImagePipeline.__name__.upper()
settings = self._generate_fake_settings(prefix=prefix)
settings = self._generate_fake_settings(tmp_path, prefix=prefix)
user_pipeline = UserDefinedImagePipeline.from_crawler(
get_crawler(None, settings)
)
@ -453,27 +447,27 @@ class TestImagesPipelineCustomSettings:
assert custom_value != self.default_pipeline_settings[pipe_attr]
assert getattr(user_pipeline, pipe_attr.lower()) == custom_value
def test_custom_settings_and_class_attrs_for_subclasses(self):
def test_custom_settings_and_class_attrs_for_subclasses(self, tmp_path):
"""
If there are custom settings for subclass AND class attributes
setting keys are preferred and override attributes.
"""
pipeline_cls = self._generate_fake_pipeline_subclass()
prefix = pipeline_cls.__name__.upper()
settings = self._generate_fake_settings(prefix=prefix)
settings = self._generate_fake_settings(tmp_path, prefix=prefix)
user_pipeline = pipeline_cls.from_crawler(get_crawler(None, settings))
for pipe_attr, settings_attr in self.img_cls_attribute_names:
custom_value = settings.get(prefix + "_" + settings_attr)
assert custom_value != self.default_pipeline_settings[pipe_attr]
assert getattr(user_pipeline, pipe_attr.lower()) == custom_value
def test_cls_attrs_with_DEFAULT_prefix(self):
def test_cls_attrs_with_DEFAULT_prefix(self, tmp_path):
class UserDefinedImagePipeline(ImagesPipeline):
DEFAULT_IMAGES_URLS_FIELD = "something"
DEFAULT_IMAGES_RESULT_FIELD = "something_else"
pipeline = UserDefinedImagePipeline.from_crawler(
get_crawler(None, {"IMAGES_STORE": self.tempdir})
get_crawler(None, {"IMAGES_STORE": tmp_path})
)
assert (
pipeline.images_result_field
@ -484,12 +478,12 @@ class TestImagesPipelineCustomSettings:
== UserDefinedImagePipeline.DEFAULT_IMAGES_URLS_FIELD
)
def test_user_defined_subclass_default_key_names(self):
def test_user_defined_subclass_default_key_names(self, tmp_path):
"""Test situation when user defines subclass of ImagePipeline,
but uses attribute names for default pipeline (without prefixing
them with pipeline class name).
"""
settings = self._generate_fake_settings()
settings = self._generate_fake_settings(tmp_path)
class UserPipe(ImagesPipeline):
pass

View File

@ -26,7 +26,7 @@ from scrapy.spiders import (
XMLFeedSpider,
)
from scrapy.spiders.init import InitSpider
from scrapy.utils.test import get_crawler
from scrapy.utils.test import get_crawler, get_reactor_settings
from tests import get_testdata, tests_datadir
@ -101,7 +101,11 @@ class TestSpider(unittest.TestCase):
@inlineCallbacks
def test_settings_in_from_crawler(self):
spider_settings = {"TEST1": "spider", "TEST2": "spider"}
project_settings = {"TEST1": "project", "TEST3": "project"}
project_settings = {
"TEST1": "project",
"TEST3": "project",
**get_reactor_settings(),
}
class TestSpider(self.spider_class):
name = "test"

View File

@ -130,9 +130,7 @@ class PickleFifoDiskQueueTest(t.FifoDiskQueueTest, FifoDiskQueueTestMixin):
) as exc_info:
q.push(sel)
assert isinstance(exc_info.value.__context__, TypeError)
# This seems to help with https://github.com/scrapy/queuelib/issues/70.
# It will need to remain under a queuelib version check after that bug is fixed.
del exc_info
q.close()
class ChunkSize1PickleFifoDiskQueueTest(PickleFifoDiskQueueTest):

View File

@ -2,6 +2,7 @@ import asyncio
import warnings
import pytest
from twisted.trial.unittest import TestCase
from scrapy.utils.defer import deferred_f_from_coro_f
from scrapy.utils.reactor import (
@ -12,10 +13,10 @@ from scrapy.utils.reactor import (
@pytest.mark.usefixtures("reactor_pytest")
class TestAsyncio:
class TestAsyncio(TestCase):
def test_is_asyncio_reactor_installed(self):
# the result should depend only on the pytest --reactor argument
assert is_asyncio_reactor_installed() == (self.reactor_pytest == "asyncio")
assert is_asyncio_reactor_installed() == (self.reactor_pytest != "default")
def test_install_asyncio_reactor(self):
from twisted.internet import reactor as original_reactor

View File

@ -1,24 +1,14 @@
from pathlib import Path
from shutil import rmtree
from tempfile import mkdtemp
from scrapy.utils.template import render_templatefile
class TestUtilsRenderTemplateFile:
def setup_method(self):
self.tmp_path = mkdtemp()
def teardown_method(self):
rmtree(self.tmp_path)
def test_simple_render(self):
def test_simple_render(self, tmp_path):
context = {"project_name": "proj", "name": "spi", "classname": "TheSpider"}
template = "from ${project_name}.spiders.${name} import ${classname}"
rendered = "from proj.spiders.spi import TheSpider"
template_path = Path(self.tmp_path, "templ.py.tmpl")
render_path = Path(self.tmp_path, "templ.py")
template_path = tmp_path / "templ.py.tmpl"
render_path = tmp_path / "templ.py"
template_path.write_text(template, encoding="utf8")
assert template_path.is_file() # Failure of test itself

78
tox.ini
View File

@ -26,7 +26,7 @@ deps =
{[test-requirements]deps}
# mitmproxy does not support PyPy
mitmproxy; implementation_name != 'pypy'
mitmproxy; implementation_name != "pypy"
setenv =
COVERAGE_CORE=sysmon
passenv =
@ -58,7 +58,7 @@ deps =
pytest >= 8.2.0
w3lib >= 2.2.0
commands =
mypy {posargs: scrapy tests}
mypy {posargs:scrapy tests}
[testenv:typing-tests]
basepython = python3.9
@ -67,7 +67,7 @@ deps =
{[testenv:typing]deps}
pytest-mypy-testing==0.1.3
commands =
pytest {posargs: tests_typing}
pytest {posargs:tests_typing}
[testenv:pre-commit]
basepython = python3
@ -96,19 +96,18 @@ commands =
[pinned]
basepython = python3.9
deps =
Protego==0.1.15
Twisted==21.7.0
cryptography==37.0.0
cssselect==0.9.1
h2==3.0
itemadapter==0.1.0
lxml==4.6.0
parsel==1.5.0
Protego==0.1.15
pyOpenSSL==22.0.0
queuelib==1.4.2
service_identity==18.1.0
Twisted[http2]==21.7.0
w3lib==1.17.0
zope.interface==5.1.0
lxml==4.6.0
{[test-requirements]deps}
# mitmproxy 8.0.0 requires upgrading some of the pinned dependencies
@ -119,7 +118,7 @@ install_command =
python -I -m pip install {opts} {packages}
commands =
; tests for docs fail with parsel < 1.8.0
pytest --cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= {posargs:--durations=10 scrapy tests}
pytest {posargs:--cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= --durations=10 scrapy tests}
[testenv:pinned]
basepython = {[pinned]basepython}
@ -131,60 +130,50 @@ setenv =
{[pinned]setenv}
commands = {[pinned]commands}
[testenv:windows-pinned]
basepython = {[pinned]basepython}
deps =
{[pinned]deps}
PyDispatcher==2.0.5
install_command = {[pinned]install_command}
setenv =
{[pinned]setenv}
commands = {[pinned]commands}
[testenv:extra-deps]
basepython = python3
deps =
{[testenv]deps}
boto3
google-cloud-storage
robotexclusionrulesparser
Pillow
Twisted[http2]
uvloop; platform_system != "Windows"
boto3
bpython # optional for shell wrapper tests
brotli; implementation_name != 'pypy' # optional for HTTP compress downloader middleware tests
brotlicffi; implementation_name == 'pypy' # optional for HTTP compress downloader middleware tests
zstandard; implementation_name != 'pypy' # optional for HTTP compress downloader middleware tests
brotli; implementation_name != "pypy" # optional for HTTP compress downloader middleware tests
brotlicffi; implementation_name == "pypy" # optional for HTTP compress downloader middleware tests
google-cloud-storage
ipython
robotexclusionrulesparser
uvloop; platform_system != "Windows"
zstandard; implementation_name != "pypy" # optional for HTTP compress downloader middleware tests
[testenv:extra-deps-pinned]
basepython = {[pinned]basepython}
deps =
{[pinned]deps}
Pillow==8.0.0
boto3==1.20.0
google-cloud-storage==1.29.0
Pillow==7.1.0
robotexclusionrulesparser==1.6.2
brotlipy
uvloop==0.14.0; platform_system != "Windows"
bpython==0.7.1
zstandard==0.1; implementation_name != 'pypy'
brotli==0.5.2; implementation_name != "pypy"
brotlicffi==0.8.0; implementation_name == "pypy"
brotlipy
google-cloud-storage==1.29.0
ipython==2.0.0
brotli==0.5.2; implementation_name != 'pypy'
brotlicffi==0.8.0; implementation_name == 'pypy'
robotexclusionrulesparser==1.6.2
uvloop==0.14.0; platform_system != "Windows"
zstandard==0.1; implementation_name != "pypy"
install_command = {[pinned]install_command}
setenv =
{[pinned]setenv}
commands = {[pinned]commands}
[testenv:asyncio]
[testenv:default-reactor]
commands =
{[testenv]commands} --reactor=asyncio
{[testenv]commands} --reactor=default
[testenv:asyncio-pinned]
[testenv:default-reactor-pinned]
basepython = {[pinned]basepython}
deps = {[testenv:pinned]deps}
commands = {[pinned]commands} --reactor=asyncio
commands = {[pinned]commands} --reactor=default
install_command = {[pinned]install_command}
setenv =
{[pinned]setenv}
@ -204,21 +193,20 @@ commands = {[testenv:pypy3]commands}
[testenv:pypy3-pinned]
basepython = pypy3.10
deps =
PyPyDispatcher==2.1.0
{[test-requirements]deps}
Protego==0.1.15
Twisted==21.7.0
cryptography==41.0.5
cssselect==0.9.1
h2==3.1
itemadapter==0.1.0
lxml==4.6.0
parsel==1.5.0
Protego==0.1.15
pyOpenSSL==23.3.0
queuelib==1.4.2
service_identity==18.1.0
Twisted[http2]==21.7.0
w3lib==1.17.0
zope.interface==5.1.0
lxml==4.6.0
{[test-requirements]deps}
PyPyDispatcher==2.1.0
commands =
; disabling both coverage and docs tests
pytest {posargs:--durations=10 scrapy tests}
@ -266,7 +254,7 @@ deps =
{[testenv]deps}
botocore>=1.4.87
commands =
pytest --cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= {posargs:tests -m requires_botocore}
pytest {posargs:--cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= tests -m requires_botocore}
[testenv:botocore-pinned]
basepython = {[pinned]basepython}
@ -277,4 +265,4 @@ install_command = {[pinned]install_command}
setenv =
{[pinned]setenv}
commands =
pytest --cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= {posargs:tests -m requires_botocore}
pytest {posargs:--cov-config=pyproject.toml --cov=scrapy --cov-report=xml --cov-report= tests -m requires_botocore}