mirror of https://github.com/scrapy/scrapy.git
Add an item_processor feed option
This commit is contained in:
parent
0c89e87b18
commit
d00dc2b8b4
|
|
@ -338,6 +338,35 @@ ItemFilter
|
||||||
:members:
|
:members:
|
||||||
|
|
||||||
|
|
||||||
|
.. _item-processor:
|
||||||
|
|
||||||
|
Item processing
|
||||||
|
===============
|
||||||
|
|
||||||
|
.. versionadded:: VERSION
|
||||||
|
|
||||||
|
The ``item_processor`` :ref:`feed option <feed-options>` takes a callable, or
|
||||||
|
its import path, that receives an accepted item and returns an iterable of the
|
||||||
|
items to export in its place:
|
||||||
|
|
||||||
|
.. code-block:: python
|
||||||
|
|
||||||
|
def split_variants(item):
|
||||||
|
for variant in item["variants"]:
|
||||||
|
yield {**item, "variants": None, **variant}
|
||||||
|
|
||||||
|
Returning an empty iterable drops the item from that feed, and returning more
|
||||||
|
than one item writes one entry per returned item.
|
||||||
|
|
||||||
|
Item processors run after :ref:`item filtering <item-filter>`, and only affect
|
||||||
|
the feed that declares them. Items are exported as returned, so the
|
||||||
|
:signal:`item_scraped` signal, :ref:`item pipelines <topics-item-pipeline>` and
|
||||||
|
the ``item_scraped_count`` stat still see the item as scraped; use item
|
||||||
|
processors for output formatting, and item pipelines for anything that should
|
||||||
|
apply to the item itself. The number of exported entries per feed is reported
|
||||||
|
as the ``feedexport/item_count/<storage class>`` stat.
|
||||||
|
|
||||||
|
|
||||||
.. _post-processing:
|
.. _post-processing:
|
||||||
|
|
||||||
Post-Processing
|
Post-Processing
|
||||||
|
|
@ -490,6 +519,13 @@ as a fallback value if that key is not provided for a specific feed definition:
|
||||||
|
|
||||||
:class:`~scrapy.extensions.feedexport.ItemFilter` is used be default.
|
:class:`~scrapy.extensions.feedexport.ItemFilter` is used be default.
|
||||||
|
|
||||||
|
- ``item_processor``: an :ref:`item processor <item-processor>` to reshape
|
||||||
|
items before they are exported.
|
||||||
|
|
||||||
|
.. versionadded:: VERSION
|
||||||
|
|
||||||
|
If undefined, items are exported as scraped.
|
||||||
|
|
||||||
- ``indent``: falls back to :setting:`FEED_EXPORT_INDENT`.
|
- ``indent``: falls back to :setting:`FEED_EXPORT_INDENT`.
|
||||||
|
|
||||||
- ``item_export_kwargs``: :class:`dict` with keyword arguments for the corresponding :ref:`item exporter class <topics-exporters>`.
|
- ``item_export_kwargs``: :class:`dict` with keyword arguments for the corresponding :ref:`item exporter class <topics-exporters>`.
|
||||||
|
|
|
||||||
|
|
@ -13,7 +13,7 @@ import re
|
||||||
import sys
|
import sys
|
||||||
import warnings
|
import warnings
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from collections.abc import Callable
|
from collections.abc import Callable, Iterable
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
from pathlib import Path, PureWindowsPath
|
from pathlib import Path, PureWindowsPath
|
||||||
from tempfile import NamedTemporaryFile
|
from tempfile import NamedTemporaryFile
|
||||||
|
|
@ -79,6 +79,8 @@ UriParamsCallableT: TypeAlias = Callable[
|
||||||
[dict[str, Any], Spider], dict[str, Any] | None
|
[dict[str, Any], Spider], dict[str, Any] | None
|
||||||
]
|
]
|
||||||
|
|
||||||
|
_ItemProcessor: TypeAlias = Callable[[Any], Iterable[Any]]
|
||||||
|
|
||||||
|
|
||||||
class ItemFilter:
|
class ItemFilter:
|
||||||
"""
|
"""
|
||||||
|
|
@ -486,6 +488,7 @@ class FeedExporter:
|
||||||
self.feeds = {}
|
self.feeds = {}
|
||||||
self.slots: list[FeedSlot] = []
|
self.slots: list[FeedSlot] = []
|
||||||
self.filters: dict[str, ItemFilter] = {}
|
self.filters: dict[str, ItemFilter] = {}
|
||||||
|
self.processors: dict[str, _ItemProcessor | None] = {}
|
||||||
self._pending_close_tasks: list[asyncio.Task[None] | Deferred[None]] = []
|
self._pending_close_tasks: list[asyncio.Task[None] | Deferred[None]] = []
|
||||||
|
|
||||||
if not self.settings["FEEDS"] and not self.settings["FEED_URI"]:
|
if not self.settings["FEEDS"] and not self.settings["FEED_URI"]:
|
||||||
|
|
@ -507,6 +510,7 @@ class FeedExporter:
|
||||||
feed_options, self.settings
|
feed_options, self.settings
|
||||||
)
|
)
|
||||||
self.filters[uri] = self._load_filter(feed_options)
|
self.filters[uri] = self._load_filter(feed_options)
|
||||||
|
self.processors[uri] = self._load_processor(feed_options)
|
||||||
# End: Backward compatibility for FEED_URI and FEED_FORMAT settings
|
# End: Backward compatibility for FEED_URI and FEED_FORMAT settings
|
||||||
|
|
||||||
# 'FEEDS' setting takes precedence over 'FEED_URI'
|
# 'FEEDS' setting takes precedence over 'FEED_URI'
|
||||||
|
|
@ -521,6 +525,7 @@ class FeedExporter:
|
||||||
feed_options, self.settings
|
feed_options, self.settings
|
||||||
)
|
)
|
||||||
self.filters[uri] = self._load_filter(feed_options)
|
self.filters[uri] = self._load_filter(feed_options)
|
||||||
|
self.processors[uri] = self._load_processor(feed_options)
|
||||||
|
|
||||||
self.storages: dict[str, type[FeedStorageProtocol]] = self._load_components(
|
self.storages: dict[str, type[FeedStorageProtocol]] = self._load_components(
|
||||||
"FEED_STORAGES"
|
"FEED_STORAGES"
|
||||||
|
|
@ -612,6 +617,9 @@ class FeedExporter:
|
||||||
logmsg = f"{slot.format} feed ({slot.itemcount} items) in: {slot.uri}"
|
logmsg = f"{slot.format} feed ({slot.itemcount} items) in: {slot.uri}"
|
||||||
slot_type = type(slot.storage).__name__
|
slot_type = type(slot.storage).__name__
|
||||||
assert self.crawler.stats
|
assert self.crawler.stats
|
||||||
|
self.crawler.stats.inc_value(
|
||||||
|
f"feedexport/item_count/{slot_type}", slot.itemcount
|
||||||
|
)
|
||||||
try:
|
try:
|
||||||
await ensure_awaitable(slot.storage.store(self._get_file(slot)))
|
await ensure_awaitable(slot.storage.store(self._get_file(slot)))
|
||||||
except Exception:
|
except Exception:
|
||||||
|
|
@ -672,30 +680,30 @@ class FeedExporter:
|
||||||
) # if slot doesn't accept item, continue with next slot
|
) # if slot doesn't accept item, continue with next slot
|
||||||
continue
|
continue
|
||||||
|
|
||||||
slot.start_exporting()
|
processor = self.processors[slot.uri_template]
|
||||||
assert slot.exporter
|
for exported_item in processor(item) if processor else (item,):
|
||||||
slot.exporter.export_item(item)
|
slot.start_exporting()
|
||||||
slot.itemcount += 1
|
assert slot.exporter
|
||||||
# create new slot for each slot with itemcount == FEED_EXPORT_BATCH_ITEM_COUNT and close the old one
|
slot.exporter.export_item(exported_item)
|
||||||
if (
|
slot.itemcount += 1
|
||||||
self.feeds[slot.uri_template]["batch_item_count"]
|
# create new slot for each slot with itemcount == FEED_EXPORT_BATCH_ITEM_COUNT and close the old one
|
||||||
and slot.itemcount >= self.feeds[slot.uri_template]["batch_item_count"]
|
if (
|
||||||
):
|
self.feeds[slot.uri_template]["batch_item_count"]
|
||||||
uri_params = self._get_uri_params(
|
and slot.itemcount
|
||||||
spider, self.feeds[slot.uri_template]["uri_params"], slot
|
>= self.feeds[slot.uri_template]["batch_item_count"]
|
||||||
)
|
):
|
||||||
self._schedule_slot_close(slot, spider)
|
uri_params = self._get_uri_params(
|
||||||
slots.append(
|
spider, self.feeds[slot.uri_template]["uri_params"], slot
|
||||||
self._start_new_batch(
|
)
|
||||||
|
self._schedule_slot_close(slot, spider)
|
||||||
|
slot = self._start_new_batch( # noqa: PLW2901
|
||||||
batch_id=slot.batch_id + 1,
|
batch_id=slot.batch_id + 1,
|
||||||
uri=apply_uri_params(slot.uri_template, uri_params),
|
uri=apply_uri_params(slot.uri_template, uri_params),
|
||||||
feed_options=self.feeds[slot.uri_template],
|
feed_options=self.feeds[slot.uri_template],
|
||||||
spider=spider,
|
spider=spider,
|
||||||
uri_template=slot.uri_template,
|
uri_template=slot.uri_template,
|
||||||
)
|
)
|
||||||
)
|
slots.append(slot)
|
||||||
else:
|
|
||||||
slots.append(slot)
|
|
||||||
self.slots = slots
|
self.slots = slots
|
||||||
|
|
||||||
def _load_components(self, setting_prefix: str) -> dict[str, Any]:
|
def _load_components(self, setting_prefix: str) -> dict[str, Any]:
|
||||||
|
|
@ -781,6 +789,10 @@ class FeedExporter:
|
||||||
)
|
)
|
||||||
return item_filter_class(feed_options)
|
return item_filter_class(feed_options)
|
||||||
|
|
||||||
|
def _load_processor(self, feed_options: dict[str, Any]) -> _ItemProcessor | None:
|
||||||
|
item_processor = feed_options.get("item_processor")
|
||||||
|
return load_object(item_processor) if item_processor else None
|
||||||
|
|
||||||
|
|
||||||
def __getattr__(name: str) -> Any: # pragma: no cover
|
def __getattr__(name: str) -> Any: # pragma: no cover
|
||||||
if name == "IFeedStorage":
|
if name == "IFeedStorage":
|
||||||
|
|
|
||||||
|
|
@ -140,6 +140,15 @@ class ExceptionJsonItemExporter(JsonItemExporter):
|
||||||
raise RuntimeError("foo")
|
raise RuntimeError("foo")
|
||||||
|
|
||||||
|
|
||||||
|
def split_foo(item):
|
||||||
|
for value in item["foo"].split(","):
|
||||||
|
yield {"foo": value}
|
||||||
|
|
||||||
|
|
||||||
|
def drop_item(item):
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
class TestFeedExport(TestFeedExportBase):
|
class TestFeedExport(TestFeedExportBase):
|
||||||
async def run_and_export(
|
async def run_and_export(
|
||||||
self, spider_cls: type[Spider], settings: dict[str, Any]
|
self, spider_cls: type[Spider], settings: dict[str, Any]
|
||||||
|
|
@ -769,6 +778,72 @@ class TestFeedExport(TestFeedExportBase):
|
||||||
for fmt, expected in formats.items():
|
for fmt, expected in formats.items():
|
||||||
assert data[fmt] == expected
|
assert data[fmt] == expected
|
||||||
|
|
||||||
|
@coroutine_test
|
||||||
|
async def test_export_based_on_item_processors(self):
|
||||||
|
items = [
|
||||||
|
MyItem({"foo": "bar1,bar2"}),
|
||||||
|
{"foo": "bar3"},
|
||||||
|
]
|
||||||
|
|
||||||
|
formats = {
|
||||||
|
"jsonlines": b'{"foo": "bar1"}\n{"foo": "bar2"}\n{"foo": "bar3"}\n',
|
||||||
|
"json": b'[\n{"foo": "bar1"},\n{"foo": "bar2"},\n{"foo": "bar3"}\n]',
|
||||||
|
"xml": (
|
||||||
|
b'<?xml version="1.0" encoding="utf-8"?>\n<items>\n'
|
||||||
|
b"<item><foo>bar1</foo></item>\n<item><foo>bar2</foo></item>\n</items>"
|
||||||
|
),
|
||||||
|
"csv": b"",
|
||||||
|
}
|
||||||
|
|
||||||
|
settings = {
|
||||||
|
"FEEDS": {
|
||||||
|
self._random_temp_filename(): {
|
||||||
|
"format": "jsonlines",
|
||||||
|
"item_processor": split_foo,
|
||||||
|
},
|
||||||
|
self._random_temp_filename(): {
|
||||||
|
"format": "json",
|
||||||
|
"item_processor": "tests.test_feedexport.split_foo",
|
||||||
|
},
|
||||||
|
self._random_temp_filename(): {
|
||||||
|
"format": "xml",
|
||||||
|
"item_classes": [MyItem],
|
||||||
|
"item_processor": split_foo,
|
||||||
|
},
|
||||||
|
self._random_temp_filename(): {
|
||||||
|
"format": "csv",
|
||||||
|
"item_processor": drop_item,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
data = await self.exported_data(items, settings)
|
||||||
|
for fmt, expected in formats.items():
|
||||||
|
assert data[fmt] == expected
|
||||||
|
|
||||||
|
@coroutine_test
|
||||||
|
async def test_item_processor_stats(self):
|
||||||
|
class TestSpider(scrapy.Spider):
|
||||||
|
name = "testspider"
|
||||||
|
start_urls = [self.mockserver.url("/")]
|
||||||
|
|
||||||
|
def parse(self, response):
|
||||||
|
yield {"foo": "bar1,bar2"}
|
||||||
|
|
||||||
|
settings = {
|
||||||
|
"FEEDS": {
|
||||||
|
path_to_url(self._random_temp_filename()): {
|
||||||
|
"format": "jsonlines",
|
||||||
|
"item_processor": split_foo,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
crawler = get_crawler(TestSpider, settings)
|
||||||
|
await crawler.crawl_async()
|
||||||
|
|
||||||
|
assert crawler.stats.get_value("item_scraped_count") == 1
|
||||||
|
assert crawler.stats.get_value("feedexport/item_count/FileFeedStorage") == 2
|
||||||
|
|
||||||
@coroutine_test
|
@coroutine_test
|
||||||
async def test_export_dicts(self):
|
async def test_export_dicts(self):
|
||||||
# When dicts are used, only keys from the first row are used as
|
# When dicts are used, only keys from the first row are used as
|
||||||
|
|
|
||||||
|
|
@ -383,6 +383,36 @@ class TestBatchDeliveries(TestFeedExportBase):
|
||||||
for expected_batch, got_batch in zip(expected, data[fmt], strict=True):
|
for expected_batch, got_batch in zip(expected, data[fmt], strict=True):
|
||||||
assert got_batch == expected_batch
|
assert got_batch == expected_batch
|
||||||
|
|
||||||
|
@coroutine_test
|
||||||
|
async def test_batch_item_count_with_item_processor(self):
|
||||||
|
def split_foo(item):
|
||||||
|
for value in item["foo"].split(","):
|
||||||
|
yield {"foo": value}
|
||||||
|
|
||||||
|
items = [{"foo": "FOO,FOO1"}, {"foo": "FOO2"}]
|
||||||
|
formats = {
|
||||||
|
"json": [
|
||||||
|
b'[{"foo": "FOO"}]',
|
||||||
|
b'[{"foo": "FOO1"}]',
|
||||||
|
b'[{"foo": "FOO2"}]',
|
||||||
|
],
|
||||||
|
}
|
||||||
|
settings = {
|
||||||
|
"FEEDS": {
|
||||||
|
self._random_temp_filename() / "json" / self._file_mark: {
|
||||||
|
"format": "json",
|
||||||
|
"indent": None,
|
||||||
|
"encoding": "utf-8",
|
||||||
|
"batch_item_count": 1,
|
||||||
|
"item_processor": split_foo,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
data = await self.exported_data(items, settings)
|
||||||
|
for fmt, expected in formats.items():
|
||||||
|
for expected_batch, got_batch in zip(expected, data[fmt], strict=True):
|
||||||
|
assert got_batch == expected_batch
|
||||||
|
|
||||||
@coroutine_test
|
@coroutine_test
|
||||||
async def test_batch_path_differ(self):
|
async def test_batch_path_differ(self):
|
||||||
"""
|
"""
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue