Document how to customize media pipeline file names from responses (#7909)

This commit is contained in:
Adrian 2026-08-09 20:08:59 +02:00 committed by GitHub
parent 5270f3cf99
commit ae68786210
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
2 changed files with 93 additions and 6 deletions

View File

@ -178,6 +178,37 @@ By overriding ``file_path`` like this:
For more information about the ``file_path`` method, see :ref:`topics-media-pipeline-override`.
.. _file-naming-response:
Naming files after the response
-------------------------------
``file_path`` also receives the ``response``, which allows naming files after
response data. For example, to determine the file extension from the
``Content-Type`` header, for URLs that do not end in a file name:
.. code-block:: python
import mimetypes
from scrapy.pipelines.files import FilesPipeline
class ContentTypeFilesPipeline(FilesPipeline):
def file_path(self, request, response=None, info=None, *, item=None):
path = super().file_path(request, response, info, item=item)
if response is None:
return path
content_type = response.headers["Content-Type"].decode()
return path + (mimetypes.guess_extension(content_type) or "")
This requires setting :setting:`FILES_EXPIRES` to ``0``. To find out whether a
file has already been downloaded, Scrapy calls ``file_path`` before the
download, with ``response`` set to ``None``, and checks the age of the file at
the resulting path. A path that depends on the response can never match that
check, and :setting:`FILES_EXPIRES` set to ``0`` disables it, at the cost of
downloading every file on every run.
.. _topics-supported-storage:
Supported Storage
@ -543,7 +574,7 @@ See here the methods that you can override in your custom Files Pipeline:
return "files/" + PurePosixPath(urlparse_cached(request).path).name
Similarly, you can use the ``item`` to determine the file path based on some item
property.
property, or the ``response``, see :ref:`file-naming-response`.
By default the :meth:`file_path` method returns
``full/<request URL hash>.<extension>``.
@ -693,7 +724,7 @@ See here the methods that you can override in your custom Images Pipeline:
return "files/" + PurePosixPath(urlparse_cached(request).path).name
Similarly, you can use the ``item`` to determine the file path based on some item
property.
property, or the ``response``, see :ref:`file-naming-response`.
By default the :meth:`file_path` method returns
``full/<request URL hash>.<extension>``.

View File

@ -1,6 +1,7 @@
import base64
import dataclasses
import logging
import mimetypes
import random
import re
import time
@ -96,8 +97,14 @@ class TestFilesPipeline:
def teardown_method(self):
rmtree(self.tempdir)
def _create_pipeline(self, pipeline_cls: type[FilesPipeline]) -> FilesPipeline:
crawler = get_crawler(DefaultSpider, {"FILES_STORE": self.tempdir})
def _create_pipeline(
self,
pipeline_cls: type[FilesPipeline],
settings: dict[str, Any] | None = None,
) -> FilesPipeline:
crawler = get_crawler(
DefaultSpider, {"FILES_STORE": self.tempdir, **(settings or {})}
)
crawler.spider = crawler._create_spider()
crawler.engine = MagicMock(download_async=mocked_download_func)
pipeline = pipeline_cls.from_crawler(crawler)
@ -394,6 +401,47 @@ class TestFilesPipeline:
request = Request("http://example.com")
assert file_path(request, item=item) == "full/path-to-store-file"
@coroutine_test
async def test_file_path_from_response(self) -> None:
"""file_path() may build the path out of the response, e.g. to get the
file extension from a response header, as long as FILES_EXPIRES is 0 to
disable the up-to-date check, which runs before the download and hence
cannot reach the same path."""
class ContentTypeFilesPipeline(FilesPipeline):
def file_path(self, request, response=None, info=None, *, item=None):
path = super().file_path(request, response, info, item=item)
if response is None:
return path
content_type = response.headers["Content-Type"].decode()
return path + (mimetypes.guess_extension(content_type) or "")
item_url = "http://example.com/download?id=1"
item = _create_item_with_files(item_url)
pipeline = self._create_pipeline(ContentTypeFilesPipeline, {"FILES_EXPIRES": 0})
request = _prepare_request_object(
item_url, headers={"Content-Type": "application/pdf"}
)
with (
mock.patch.object(FilesPipeline, "inc_stats", return_value=True),
# A fresh file at the response-less path is ignored thanks to
# FILES_EXPIRES being 0.
mock.patch.object(
FSFilesStore,
"stat_file",
return_value={"checksum": "abc", "last_modified": time.time()},
),
mock.patch.object(
FilesPipeline, "get_media_requests", return_value=[request]
),
):
result = await pipeline.process_item(item)
file_info = result["files"][0]
assert file_info["status"] == "downloaded"
assert file_info["path"].endswith(".pdf")
assert (Path(self.tempdir) / file_info["path"]).read_bytes() == b"data"
def test_media_failed_filtered_request(
self, caplog: pytest.LogCaptureFixture
) -> None:
@ -1133,10 +1181,18 @@ def _create_item_with_files(*files: str) -> ItemWithFiles:
return item
def _prepare_request_object(item_url: str, flags: list[str] | None = None) -> Request:
def _prepare_request_object(
item_url: str,
flags: list[str] | None = None,
headers: dict[str, str] | None = None,
) -> Request:
return Request(
item_url,
meta={"response": Response(item_url, status=200, body=b"data", flags=flags)},
meta={
"response": Response(
item_url, status=200, body=b"data", flags=flags, headers=headers
)
},
)