diff --git a/docs/topics/media-pipeline.rst b/docs/topics/media-pipeline.rst index b16066d0c..576feae7e 100644 --- a/docs/topics/media-pipeline.rst +++ b/docs/topics/media-pipeline.rst @@ -178,6 +178,37 @@ By overriding ``file_path`` like this: For more information about the ``file_path`` method, see :ref:`topics-media-pipeline-override`. +.. _file-naming-response: + +Naming files after the response +------------------------------- + +``file_path`` also receives the ``response``, which allows naming files after +response data. For example, to determine the file extension from the +``Content-Type`` header, for URLs that do not end in a file name: + +.. code-block:: python + + import mimetypes + + from scrapy.pipelines.files import FilesPipeline + + + class ContentTypeFilesPipeline(FilesPipeline): + def file_path(self, request, response=None, info=None, *, item=None): + path = super().file_path(request, response, info, item=item) + if response is None: + return path + content_type = response.headers["Content-Type"].decode() + return path + (mimetypes.guess_extension(content_type) or "") + +This requires setting :setting:`FILES_EXPIRES` to ``0``. To find out whether a +file has already been downloaded, Scrapy calls ``file_path`` before the +download, with ``response`` set to ``None``, and checks the age of the file at +the resulting path. A path that depends on the response can never match that +check, and :setting:`FILES_EXPIRES` set to ``0`` disables it, at the cost of +downloading every file on every run. + .. _topics-supported-storage: Supported Storage @@ -543,7 +574,7 @@ See here the methods that you can override in your custom Files Pipeline: return "files/" + PurePosixPath(urlparse_cached(request).path).name Similarly, you can use the ``item`` to determine the file path based on some item - property. + property, or the ``response``, see :ref:`file-naming-response`. By default the :meth:`file_path` method returns ``full/.``. @@ -693,7 +724,7 @@ See here the methods that you can override in your custom Images Pipeline: return "files/" + PurePosixPath(urlparse_cached(request).path).name Similarly, you can use the ``item`` to determine the file path based on some item - property. + property, or the ``response``, see :ref:`file-naming-response`. By default the :meth:`file_path` method returns ``full/.``. diff --git a/tests/test_pipeline_files.py b/tests/test_pipeline_files.py index 3a8a3d0f7..f40619933 100644 --- a/tests/test_pipeline_files.py +++ b/tests/test_pipeline_files.py @@ -1,6 +1,7 @@ import base64 import dataclasses import logging +import mimetypes import random import re import time @@ -96,8 +97,14 @@ class TestFilesPipeline: def teardown_method(self): rmtree(self.tempdir) - def _create_pipeline(self, pipeline_cls: type[FilesPipeline]) -> FilesPipeline: - crawler = get_crawler(DefaultSpider, {"FILES_STORE": self.tempdir}) + def _create_pipeline( + self, + pipeline_cls: type[FilesPipeline], + settings: dict[str, Any] | None = None, + ) -> FilesPipeline: + crawler = get_crawler( + DefaultSpider, {"FILES_STORE": self.tempdir, **(settings or {})} + ) crawler.spider = crawler._create_spider() crawler.engine = MagicMock(download_async=mocked_download_func) pipeline = pipeline_cls.from_crawler(crawler) @@ -394,6 +401,47 @@ class TestFilesPipeline: request = Request("http://example.com") assert file_path(request, item=item) == "full/path-to-store-file" + @coroutine_test + async def test_file_path_from_response(self) -> None: + """file_path() may build the path out of the response, e.g. to get the + file extension from a response header, as long as FILES_EXPIRES is 0 to + disable the up-to-date check, which runs before the download and hence + cannot reach the same path.""" + + class ContentTypeFilesPipeline(FilesPipeline): + def file_path(self, request, response=None, info=None, *, item=None): + path = super().file_path(request, response, info, item=item) + if response is None: + return path + content_type = response.headers["Content-Type"].decode() + return path + (mimetypes.guess_extension(content_type) or "") + + item_url = "http://example.com/download?id=1" + item = _create_item_with_files(item_url) + pipeline = self._create_pipeline(ContentTypeFilesPipeline, {"FILES_EXPIRES": 0}) + request = _prepare_request_object( + item_url, headers={"Content-Type": "application/pdf"} + ) + with ( + mock.patch.object(FilesPipeline, "inc_stats", return_value=True), + # A fresh file at the response-less path is ignored thanks to + # FILES_EXPIRES being 0. + mock.patch.object( + FSFilesStore, + "stat_file", + return_value={"checksum": "abc", "last_modified": time.time()}, + ), + mock.patch.object( + FilesPipeline, "get_media_requests", return_value=[request] + ), + ): + result = await pipeline.process_item(item) + + file_info = result["files"][0] + assert file_info["status"] == "downloaded" + assert file_info["path"].endswith(".pdf") + assert (Path(self.tempdir) / file_info["path"]).read_bytes() == b"data" + def test_media_failed_filtered_request( self, caplog: pytest.LogCaptureFixture ) -> None: @@ -1133,10 +1181,18 @@ def _create_item_with_files(*files: str) -> ItemWithFiles: return item -def _prepare_request_object(item_url: str, flags: list[str] | None = None) -> Request: +def _prepare_request_object( + item_url: str, + flags: list[str] | None = None, + headers: dict[str, str] | None = None, +) -> Request: return Request( item_url, - meta={"response": Response(item_url, status=200, body=b"data", flags=flags)}, + meta={ + "response": Response( + item_url, status=200, body=b"data", flags=flags, headers=headers + ) + }, )