mirror of https://github.com/scrapy/scrapy.git
Document how to customize media pipeline file names from responses (#7909)
This commit is contained in:
parent
5270f3cf99
commit
ae68786210
|
|
@ -178,6 +178,37 @@ By overriding ``file_path`` like this:
|
|||
|
||||
For more information about the ``file_path`` method, see :ref:`topics-media-pipeline-override`.
|
||||
|
||||
.. _file-naming-response:
|
||||
|
||||
Naming files after the response
|
||||
-------------------------------
|
||||
|
||||
``file_path`` also receives the ``response``, which allows naming files after
|
||||
response data. For example, to determine the file extension from the
|
||||
``Content-Type`` header, for URLs that do not end in a file name:
|
||||
|
||||
.. code-block:: python
|
||||
|
||||
import mimetypes
|
||||
|
||||
from scrapy.pipelines.files import FilesPipeline
|
||||
|
||||
|
||||
class ContentTypeFilesPipeline(FilesPipeline):
|
||||
def file_path(self, request, response=None, info=None, *, item=None):
|
||||
path = super().file_path(request, response, info, item=item)
|
||||
if response is None:
|
||||
return path
|
||||
content_type = response.headers["Content-Type"].decode()
|
||||
return path + (mimetypes.guess_extension(content_type) or "")
|
||||
|
||||
This requires setting :setting:`FILES_EXPIRES` to ``0``. To find out whether a
|
||||
file has already been downloaded, Scrapy calls ``file_path`` before the
|
||||
download, with ``response`` set to ``None``, and checks the age of the file at
|
||||
the resulting path. A path that depends on the response can never match that
|
||||
check, and :setting:`FILES_EXPIRES` set to ``0`` disables it, at the cost of
|
||||
downloading every file on every run.
|
||||
|
||||
.. _topics-supported-storage:
|
||||
|
||||
Supported Storage
|
||||
|
|
@ -543,7 +574,7 @@ See here the methods that you can override in your custom Files Pipeline:
|
|||
return "files/" + PurePosixPath(urlparse_cached(request).path).name
|
||||
|
||||
Similarly, you can use the ``item`` to determine the file path based on some item
|
||||
property.
|
||||
property, or the ``response``, see :ref:`file-naming-response`.
|
||||
|
||||
By default the :meth:`file_path` method returns
|
||||
``full/<request URL hash>.<extension>``.
|
||||
|
|
@ -693,7 +724,7 @@ See here the methods that you can override in your custom Images Pipeline:
|
|||
return "files/" + PurePosixPath(urlparse_cached(request).path).name
|
||||
|
||||
Similarly, you can use the ``item`` to determine the file path based on some item
|
||||
property.
|
||||
property, or the ``response``, see :ref:`file-naming-response`.
|
||||
|
||||
By default the :meth:`file_path` method returns
|
||||
``full/<request URL hash>.<extension>``.
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import base64
|
||||
import dataclasses
|
||||
import logging
|
||||
import mimetypes
|
||||
import random
|
||||
import re
|
||||
import time
|
||||
|
|
@ -96,8 +97,14 @@ class TestFilesPipeline:
|
|||
def teardown_method(self):
|
||||
rmtree(self.tempdir)
|
||||
|
||||
def _create_pipeline(self, pipeline_cls: type[FilesPipeline]) -> FilesPipeline:
|
||||
crawler = get_crawler(DefaultSpider, {"FILES_STORE": self.tempdir})
|
||||
def _create_pipeline(
|
||||
self,
|
||||
pipeline_cls: type[FilesPipeline],
|
||||
settings: dict[str, Any] | None = None,
|
||||
) -> FilesPipeline:
|
||||
crawler = get_crawler(
|
||||
DefaultSpider, {"FILES_STORE": self.tempdir, **(settings or {})}
|
||||
)
|
||||
crawler.spider = crawler._create_spider()
|
||||
crawler.engine = MagicMock(download_async=mocked_download_func)
|
||||
pipeline = pipeline_cls.from_crawler(crawler)
|
||||
|
|
@ -394,6 +401,47 @@ class TestFilesPipeline:
|
|||
request = Request("http://example.com")
|
||||
assert file_path(request, item=item) == "full/path-to-store-file"
|
||||
|
||||
@coroutine_test
|
||||
async def test_file_path_from_response(self) -> None:
|
||||
"""file_path() may build the path out of the response, e.g. to get the
|
||||
file extension from a response header, as long as FILES_EXPIRES is 0 to
|
||||
disable the up-to-date check, which runs before the download and hence
|
||||
cannot reach the same path."""
|
||||
|
||||
class ContentTypeFilesPipeline(FilesPipeline):
|
||||
def file_path(self, request, response=None, info=None, *, item=None):
|
||||
path = super().file_path(request, response, info, item=item)
|
||||
if response is None:
|
||||
return path
|
||||
content_type = response.headers["Content-Type"].decode()
|
||||
return path + (mimetypes.guess_extension(content_type) or "")
|
||||
|
||||
item_url = "http://example.com/download?id=1"
|
||||
item = _create_item_with_files(item_url)
|
||||
pipeline = self._create_pipeline(ContentTypeFilesPipeline, {"FILES_EXPIRES": 0})
|
||||
request = _prepare_request_object(
|
||||
item_url, headers={"Content-Type": "application/pdf"}
|
||||
)
|
||||
with (
|
||||
mock.patch.object(FilesPipeline, "inc_stats", return_value=True),
|
||||
# A fresh file at the response-less path is ignored thanks to
|
||||
# FILES_EXPIRES being 0.
|
||||
mock.patch.object(
|
||||
FSFilesStore,
|
||||
"stat_file",
|
||||
return_value={"checksum": "abc", "last_modified": time.time()},
|
||||
),
|
||||
mock.patch.object(
|
||||
FilesPipeline, "get_media_requests", return_value=[request]
|
||||
),
|
||||
):
|
||||
result = await pipeline.process_item(item)
|
||||
|
||||
file_info = result["files"][0]
|
||||
assert file_info["status"] == "downloaded"
|
||||
assert file_info["path"].endswith(".pdf")
|
||||
assert (Path(self.tempdir) / file_info["path"]).read_bytes() == b"data"
|
||||
|
||||
def test_media_failed_filtered_request(
|
||||
self, caplog: pytest.LogCaptureFixture
|
||||
) -> None:
|
||||
|
|
@ -1133,10 +1181,18 @@ def _create_item_with_files(*files: str) -> ItemWithFiles:
|
|||
return item
|
||||
|
||||
|
||||
def _prepare_request_object(item_url: str, flags: list[str] | None = None) -> Request:
|
||||
def _prepare_request_object(
|
||||
item_url: str,
|
||||
flags: list[str] | None = None,
|
||||
headers: dict[str, str] | None = None,
|
||||
) -> Request:
|
||||
return Request(
|
||||
item_url,
|
||||
meta={"response": Response(item_url, status=200, body=b"data", flags=flags)},
|
||||
meta={
|
||||
"response": Response(
|
||||
item_url, status=200, body=b"data", flags=flags, headers=headers
|
||||
)
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
|
|
|
|||
Loading…
Reference in New Issue