From 0aa61cec13da3b3a9d6b0807c7c0be6153b221ef Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Wed, 14 Jul 2021 20:57:33 +0530 Subject: [PATCH 01/81] install xtractmime --- coverage.xml | 11298 ++++++++++++++++++++++++++++++++++++++ scrapy/responsetypes.py | 1 + tox.ini | 1 + 3 files changed, 11300 insertions(+) create mode 100644 coverage.xml diff --git a/coverage.xml b/coverage.xml new file mode 100644 index 000000000..c44163a6b --- /dev/null +++ b/coverage.xml @@ -0,0 +1,11298 @@ + + + + + + /Users/akshaysharma/Desktop/git/scrapy/scrapy + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 6ed9f8b8f..6ad41066f 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -5,6 +5,7 @@ based on different criteria. from mimetypes import MimeTypes from pkgutil import get_data from io import StringIO +from xtractmime import extract_mime from scrapy.http import Response from scrapy.utils.misc import load_object diff --git a/tox.ini b/tox.ini index 8167aff96..0ffc4a57f 100644 --- a/tox.ini +++ b/tox.ini @@ -17,6 +17,7 @@ deps = #mitmproxy >= 5.3.0; python_version >= '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4; python_version >= '3.7' and python_version < '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4, < 5; python_version >= '3.6' and python_version < '3.7' and platform_system != 'Windows' and implementation_name != 'pypy' + git+https://github.com/scrapy/xtractmime.git@compute-mime#egg=xtractmime # Extras botocore>=1.4.87 passenv = From c85b780184b046fd596e4e3ed06ee3fba286cbd7 Mon Sep 17 00:00:00 2001 From: Akshay Sharma <42249933+akshaysharmajs@users.noreply.github.com> Date: Wed, 14 Jul 2021 21:33:56 +0530 Subject: [PATCH 02/81] Update scrapy/responsetypes.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: Adrián Chaves --- scrapy/responsetypes.py | 1 + 1 file changed, 1 insertion(+) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 6ad41066f..9592dcd73 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -5,6 +5,7 @@ based on different criteria. from mimetypes import MimeTypes from pkgutil import get_data from io import StringIO + from xtractmime import extract_mime from scrapy.http import Response From cf19c4f5917574a0d952d82b82c2b67cb463604f Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Wed, 14 Jul 2021 21:35:23 +0530 Subject: [PATCH 03/81] Remove coverage.xml --- coverage.xml | 11298 ------------------------------------------------- 1 file changed, 11298 deletions(-) delete mode 100644 coverage.xml diff --git a/coverage.xml b/coverage.xml deleted file mode 100644 index c44163a6b..000000000 --- a/coverage.xml +++ /dev/null @@ -1,11298 +0,0 @@ - - - - - - /Users/akshaysharma/Desktop/git/scrapy/scrapy - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - From 265301215ab8d15832a426075c6268c6a5da6acf Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Thu, 15 Jul 2021 20:03:11 +0530 Subject: [PATCH 04/81] Add extract_mime --- coverage.xml | 11301 ++++++++++++++++++++++++++++++++++ scrapy/responsetypes.py | 41 +- tests/test_responsetypes.py | 15 +- tox.ini | 2 +- 4 files changed, 11336 insertions(+), 23 deletions(-) create mode 100644 coverage.xml diff --git a/coverage.xml b/coverage.xml new file mode 100644 index 000000000..018e4cfce --- /dev/null +++ b/coverage.xml @@ -0,0 +1,11301 @@ + + + + + + /Users/akshaysharma/Desktop/git/scrapy/scrapy + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 9592dcd73..b6f7b6f68 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -5,6 +5,7 @@ based on different criteria. from mimetypes import MimeTypes from pkgutil import get_data from io import StringIO +from urllib.parse import urlparse from xtractmime import extract_mime @@ -42,6 +43,9 @@ class ResponseTypes: def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" + if isinstance(mimetype, bytes): + mimetype = mimetype.decode() + if mimetype is None: return Response elif mimetype in self.classes: @@ -88,34 +92,33 @@ class ResponseTypes: else: return Response - def from_body(self, body): - """Try to guess the appropriate response based on the body content. - This method is a bit magic and could be improved in the future, but - it's not meant to be used except for special cases where response types - cannot be guess using more straightforward methods.""" - chunk = body[:5000] - chunk = to_bytes(chunk) - if not binary_is_text(chunk): - return self.from_mimetype('application/octet-stream') - elif b"" in chunk.lower(): - return self.from_mimetype('text/html') - elif b" {retcls} != {cls}" - def test_from_body(self): + """def test_from_body(self): mappings = [ (b'\x03\x02\xdf\xdd\x23', Response), (b'Some plain text\ndata with tabs\t and null bytes\0', TextResponse), @@ -59,6 +59,7 @@ class ResponseTypesTest(unittest.TestCase): for source, cls in mappings: retcls = responsetypes.from_body(source) assert retcls is cls, f"{source} ==> {retcls} != {cls}" + """ def test_from_headers(self): mappings = [ @@ -79,10 +80,18 @@ class ResponseTypesTest(unittest.TestCase): # headers takes precedence over url ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), 'url': 'http://www.example.com/item/'}, HtmlResponse), + ({'body': b'Some plain text data with tabs and null bytes', + 'url': 'http://www.example.com/item/', + 'headers': Headers({'Content-Type': ['text/html; charset=utf-8'], + 'X-Content-Type-Options': 'nosniff'})}, TextResponse), + ({'body': b'\x03\x02\xdf\xdd\x23', 'url': '://www.example.com/item/', + 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), + ({'body': b'Some plain text data with tabs and null bytes'}, TextResponse), + ({'body': b'Hello'}, HtmlResponse), + ({'body': b' Date: Thu, 15 Jul 2021 20:04:27 +0530 Subject: [PATCH 05/81] Small fix --- coverage.xml | 11301 ------------------------------------------------- tox.ini | 2 +- 2 files changed, 1 insertion(+), 11302 deletions(-) delete mode 100644 coverage.xml diff --git a/coverage.xml b/coverage.xml deleted file mode 100644 index 018e4cfce..000000000 --- a/coverage.xml +++ /dev/null @@ -1,11301 +0,0 @@ - - - - - - /Users/akshaysharma/Desktop/git/scrapy/scrapy - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - - diff --git a/tox.ini b/tox.ini index 78a1f14b1..0ffc4a57f 100644 --- a/tox.ini +++ b/tox.ini @@ -29,7 +29,7 @@ passenv = #allow tox virtualenv to upgrade pip/wheel/setuptools download = true commands = - py.test --cov=scrapy --cov-report=html --cov-report= {posargs:--durations=10 docs scrapy tests} + py.test --cov=scrapy --cov-report=xml --cov-report= {posargs:--durations=10 docs scrapy tests} install_command = pip install -U -ctests/upper-constraints.txt {opts} {packages} From 211fc62b696896085750dc6ef66b58dc17ebde84 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Thu, 15 Jul 2021 20:06:56 +0530 Subject: [PATCH 06/81] Remove commented func --- tests/test_responsetypes.py | 12 ------------ 1 file changed, 12 deletions(-) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index f8968c1c1..ed817b3ce 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -49,18 +49,6 @@ class ResponseTypesTest(unittest.TestCase): retcls = responsetypes.from_content_type(source) assert retcls is cls, f"{source} ==> {retcls} != {cls}" - """def test_from_body(self): - mappings = [ - (b'\x03\x02\xdf\xdd\x23', Response), - (b'Some plain text\ndata with tabs\t and null bytes\0', TextResponse), - (b'Hello', HtmlResponse), - (b' {retcls} != {cls}" - """ - def test_from_headers(self): mappings = [ ({'Content-Type': ['text/html; charset=utf-8']}, HtmlResponse), From 763ec07b7b6e599816879a00e2c6fe5d5e92a201 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sun, 18 Jul 2021 03:56:12 +0530 Subject: [PATCH 07/81] Deprecation Warnings --- scrapy/responsetypes.py | 65 ++++++++++++++++++++++++------------- tests/test_responsetypes.py | 21 ++++++++---- 2 files changed, 58 insertions(+), 28 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index b6f7b6f68..1c4303631 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -6,9 +6,11 @@ from mimetypes import MimeTypes from pkgutil import get_data from io import StringIO from urllib.parse import urlparse +from warnings import warn from xtractmime import extract_mime +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response from scrapy.utils.misc import load_object from scrapy.utils.python import binary_is_text, to_bytes, to_unicode @@ -43,9 +45,6 @@ class ResponseTypes: def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" - if isinstance(mimetype, bytes): - mimetype = mimetype.decode() - if mimetype is None: return Response elif mimetype in self.classes: @@ -57,12 +56,16 @@ class ResponseTypes: def from_content_type(self, content_type, content_encoding=None): """Return the most appropriate Response class from an HTTP Content-Type header """ + warn('ResponseTypes.from_content_type is deprecated, ' + 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) if content_encoding: return Response mimetype = to_unicode(content_type).split(';')[0].strip().lower() return self.from_mimetype(mimetype) def from_content_disposition(self, content_disposition): + warn('ResponseTypes.from_content_disposition is deprecated, ' + 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) try: filename = to_unicode( content_disposition, encoding='latin-1', errors='replace' @@ -74,6 +77,8 @@ class ResponseTypes: def from_headers(self, headers): """Return the most appropriate Response class by looking at the HTTP headers""" + warn('ResponseTypes.from_headers is deprecated, ' + 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) cls = Response if b'Content-Type' in headers: cls = self.from_content_type( @@ -86,39 +91,55 @@ class ResponseTypes: def from_filename(self, filename): """Return the most appropriate Response class from a file name""" + warn('ResponseTypes.from_filename is deprecated, ' + 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) mimetype, encoding = self.mimetypes.guess_type(filename) if mimetype and not encoding: return self.from_mimetype(mimetype) else: return Response + def from_body(self, body): + """Try to guess the appropriate response based on the body content. + This method is a bit magic and could be improved in the future, but + it's not meant to be used except for special cases where response types + cannot be guess using more straightforward methods.""" + warn('ResponseTypes.from_body is deprecated, ' + 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) + chunk = body[:5000] + chunk = to_bytes(chunk) + if not binary_is_text(chunk): + return self.from_mimetype('application/octet-stream') + elif b"" in chunk.lower(): + return self.from_mimetype('text/html') + elif b" {retcls} != {cls}" + def test_from_body(self): + mappings = [ + (b'\x03\x02\xdf\xdd\x23', Response), + (b'Some plain text\ndata with tabs\t and null bytes\0', TextResponse), + (b'Hello', HtmlResponse), + (b' {retcls} != {cls}" + def test_from_headers(self): mappings = [ ({'Content-Type': ['text/html; charset=utf-8']}, HtmlResponse), @@ -62,16 +73,14 @@ class ResponseTypesTest(unittest.TestCase): assert retcls is cls, f"{source} ==> {retcls} != {cls}" def test_from_args(self): - # TODO: add more tests that check precedence between the different arguments mappings = [ ({'url': 'http://www.example.com/data.csv'}, TextResponse), - # headers takes precedence over url - ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), - 'url': 'http://www.example.com/item/'}, HtmlResponse), + ({'headers': Headers({'Content-Type': ['text/plain; charset=utf-8']}), + 'url': 'http://www.example.com/item/'}, TextResponse), ({'body': b'Some plain text data with tabs and null bytes', 'url': 'http://www.example.com/item/', - 'headers': Headers({'Content-Type': ['text/html; charset=utf-8'], - 'X-Content-Type-Options': 'nosniff'})}, TextResponse), + 'headers': Headers({'Content-Type': ['text/html; charset=utf-8'], })}, + TextResponse), ({'body': b'\x03\x02\xdf\xdd\x23', 'url': '://www.example.com/item/', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), ({'body': b'Some plain text data with tabs and null bytes'}, TextResponse), From 1512ed22b264a04feef5823c7661db7ab3093507 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sun, 25 Jul 2021 19:42:27 +0530 Subject: [PATCH 08/81] Pre n post xtractmime tests --- tests/test_responsetypes.py | 32 ++++++++++++++++++++++---------- 1 file changed, 22 insertions(+), 10 deletions(-) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 26d4a097d..2d8f1b082 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -72,23 +72,35 @@ class ResponseTypesTest(unittest.TestCase): retcls = responsetypes.from_headers(source) assert retcls is cls, f"{source} ==> {retcls} != {cls}" - def test_from_args(self): + def test_from_args_pre_xtractmime(self): mappings = [ ({'url': 'http://www.example.com/data.csv'}, TextResponse), - ({'headers': Headers({'Content-Type': ['text/plain; charset=utf-8']}), - 'url': 'http://www.example.com/item/'}, TextResponse), + ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), + 'url': 'http://www.example.com/item/'}, HtmlResponse), # Failing with xtractmime, returning TextResponse expected HtmlResponse + ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), + 'url': 'http://www.example.com/page/'}, Response), # Failing with xtractmime, returning TextResponse expected Response + ] + for source, cls in mappings: + retcls = responsetypes.from_args(**source) + assert retcls is cls, f"{source} ==> {retcls} != {cls}" + + def test_from_args_post_xtractmime(self): + mappings = [ ({'body': b'Some plain text data with tabs and null bytes', - 'url': 'http://www.example.com/item/', 'headers': Headers({'Content-Type': ['text/html; charset=utf-8'], })}, TextResponse), - ({'body': b'\x03\x02\xdf\xdd\x23', 'url': '://www.example.com/item/', - 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), - ({'body': b'Some plain text data with tabs and null bytes'}, TextResponse), + ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, + Response), + # different behaviour with http and non-http urls + ({'body': b'\x00\xfe\xff', 'url': 'http://www.example.com/item/', + 'headers': Headers({'Content-Type': b'text/plain'})}, Response), + ({'body': b'\x00\xfe\xff', 'url': '://www.example.com/item/', + 'headers': Headers({'Content-Type': b'text/plain'})}, TextResponse), + ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), # Failing with xtractmime, return TextResponse expected HtmlResponse + ({'body': b'Some plain text\ndata with tabs\t and null bytes\0'}, Response), # earlier expected to be binary response, refer "test_from_body()" ({'body': b'Hello'}, HtmlResponse), ({'body': b' Date: Sat, 31 Jul 2021 01:57:09 +0530 Subject: [PATCH 09/81] Add guess content type --- scrapy/responsetypes.py | 28 +++++++++++++++++++++++----- 1 file changed, 23 insertions(+), 5 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 1c4303631..67e64696d 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -8,7 +8,7 @@ from io import StringIO from urllib.parse import urlparse from warnings import warn -from xtractmime import extract_mime +from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response @@ -117,16 +117,28 @@ class ResponseTypes: else: return self.from_mimetype('text') + def __guess_content_type(self, content_disposition=None, url=None, filename=None): + if content_disposition: + filename = content_disposition.split(b';') + if len(filename) != 1: + filename = filename[-1].split(b'=')[1].strip(b'"\'') + else: + filename = filename[0] + return self.mimetypes.guess_type(filename.decode())[0] + + if url: + return self.mimetypes.guess_type(url)[0] + + if filename: + return self.mimetypes.guess_type(filename)[0] + def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on the given arguments.""" - if filename: - warn("'filename' keyword argument is deprecated, " - "please ignore this parameter as it is no longer required", - ScrapyDeprecationWarning) if not body: body = b'' + body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") cls = Response http_origin = True content_types = None @@ -136,6 +148,12 @@ class ResponseTypes: if headers and b'Content-Type' in headers: content_types = tuple(headers.getlist(b'Content-Type')) + elif headers and b'Content-Disposition' in headers: + content_types = (self.__guess_content_type(content_disposition=headers.get(b'Content-Disposition')).encode(),) + elif url: + content_types = (self.__guess_content_type(url=url).encode(),) + elif filename: + content_types = (self.__guess_content_type(filename=filename).encode(),) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) cls = self.from_mimetype(mime_type.decode()) From 180b457f771ccfbb7f2ddd8dbb9989fe51d2929c Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sat, 31 Jul 2021 17:38:28 +0530 Subject: [PATCH 10/81] Format change --- scrapy/responsetypes.py | 36 +++++++++++++++++------------------- tests/test_responsetypes.py | 4 +--- 2 files changed, 18 insertions(+), 22 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 67e64696d..c6b00c2a1 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -117,20 +117,20 @@ class ResponseTypes: else: return self.from_mimetype('text') - def __guess_content_type(self, content_disposition=None, url=None, filename=None): - if content_disposition: - filename = content_disposition.split(b';') - if len(filename) != 1: - filename = filename[-1].split(b'=')[1].strip(b'"\'') - else: - filename = filename[0] - return self.mimetypes.guess_type(filename.decode())[0] + def _guess_content_type(self, content_type=None, content_disposition=None, url=None, filename=None): + if content_type: + return tuple(content_type) - if url: - return self.mimetypes.guess_type(url)[0] + if content_disposition: + filename = content_disposition.split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() + + elif url: + filename = url if filename: - return self.mimetypes.guess_type(filename)[0] + return (self.mimetypes.guess_type(filename)[0].encode(),) + + return None def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on @@ -146,14 +146,12 @@ class ResponseTypes: if url and urlparse(url).scheme not in ("http", "https"): http_origin = False - if headers and b'Content-Type' in headers: - content_types = tuple(headers.getlist(b'Content-Type')) - elif headers and b'Content-Disposition' in headers: - content_types = (self.__guess_content_type(content_disposition=headers.get(b'Content-Disposition')).encode(),) - elif url: - content_types = (self.__guess_content_type(url=url).encode(),) - elif filename: - content_types = (self.__guess_content_type(filename=filename).encode(),) + if headers: + content_types = self._guess_content_type(content_type=headers.getlist(b'Content-Type'), + content_disposition=headers.get(b'Content-Disposition'), + url=url, filename=filename) + else: + content_types = self._guess_content_type(url=url, filename=filename) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) cls = self.from_mimetype(mime_type.decode()) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 2d8f1b082..69ee1341e 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -86,9 +86,7 @@ class ResponseTypesTest(unittest.TestCase): def test_from_args_post_xtractmime(self): mappings = [ - ({'body': b'Some plain text data with tabs and null bytes', - 'headers': Headers({'Content-Type': ['text/html; charset=utf-8'], })}, - TextResponse), + ({'body': b'Some plain text data with tabs and null bytes'}, TextResponse), ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), # different behaviour with http and non-http urls From 5ee086e7b6b732f33553a0c8b7e34c488c660a63 Mon Sep 17 00:00:00 2001 From: Akshay Sharma <42249933+akshaysharmajs@users.noreply.github.com> Date: Sat, 31 Jul 2021 20:30:43 +0530 Subject: [PATCH 11/81] Update scrapy/responsetypes.py MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Co-authored-by: Adrián Chaves --- scrapy/responsetypes.py | 15 ++------------- 1 file changed, 2 insertions(+), 13 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index c6b00c2a1..7d67675d6 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -140,19 +140,8 @@ class ResponseTypes: body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") cls = Response - http_origin = True - content_types = None - - if url and urlparse(url).scheme not in ("http", "https"): - http_origin = False - - if headers: - content_types = self._guess_content_type(content_type=headers.getlist(b'Content-Type'), - content_disposition=headers.get(b'Content-Disposition'), - url=url, filename=filename) - else: - content_types = self._guess_content_type(url=url, filename=filename) - + http_origin = not url or urlparse(url).scheme in ("http", "https") + content_types = self._guess_content_type(headers=headers, url=url, filename=filename) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) cls = self.from_mimetype(mime_type.decode()) From 3a20b79de7259eda4c7ba6486331a7aa4e370eb6 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sat, 31 Jul 2021 21:08:42 +0530 Subject: [PATCH 12/81] Move headers logic --- scrapy/responsetypes.py | 11 +++++------ tox.ini | 2 +- 2 files changed, 6 insertions(+), 7 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 7d67675d6..632787e06 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -117,13 +117,12 @@ class ResponseTypes: else: return self.from_mimetype('text') - def _guess_content_type(self, content_type=None, content_disposition=None, url=None, filename=None): - if content_type: - return tuple(content_type) - - if content_disposition: - filename = content_disposition.split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() + def _guess_content_type(self, headers=None, url=None, filename=None): + if headers and b'Content-Type' in headers: + return tuple(headers.getlist(b'Content-Type')) + if headers and b'Content-Disposition' in headers: + filename = headers.get(b'Content-Disposition').split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() elif url: filename = url diff --git a/tox.ini b/tox.ini index 0ffc4a57f..58e902ec1 100644 --- a/tox.ini +++ b/tox.ini @@ -17,7 +17,7 @@ deps = #mitmproxy >= 5.3.0; python_version >= '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4; python_version >= '3.7' and python_version < '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4, < 5; python_version >= '3.6' and python_version < '3.7' and platform_system != 'Windows' and implementation_name != 'pypy' - git+https://github.com/scrapy/xtractmime.git@compute-mime#egg=xtractmime + git+https://github.com/scrapy/xtractmime.git@html-fix#egg=xtractmime # Extras botocore>=1.4.87 passenv = From f573d11ec2a45d0aab14bc122b64242c8fe19455 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sun, 1 Aug 2021 14:40:39 +0530 Subject: [PATCH 13/81] Logic fix --- scrapy/responsetypes.py | 18 ++++++++++++++++-- tests/test_responsetypes.py | 20 ++++++++++---------- 2 files changed, 26 insertions(+), 12 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 632787e06..fd7310c80 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -9,6 +9,7 @@ from urllib.parse import urlparse from warnings import warn from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime +from xtractmime._utils import contains_binary from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response @@ -127,7 +128,11 @@ class ResponseTypes: filename = url if filename: - return (self.mimetypes.guess_type(filename)[0].encode(),) + mimetype, encoding = self.mimetypes.guess_type(filename) + if encoding: + return (f"application/{encoding}".encode(),) + else: + return (mimetype.encode(),) return None @@ -137,7 +142,16 @@ class ResponseTypes: if not body: body = b'' - body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") + contains_binary_bytes = False + + for index in range(len(body)): + if body[index:index + 1] != b"\x00" and contains_binary(body[index:index + 1]): + contains_binary_bytes = True + break + + if not contains_binary_bytes: + body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") + cls = Response http_origin = not url or urlparse(url).scheme in ("http", "https") content_types = self._guess_content_type(headers=headers, url=url, filename=filename) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 69ee1341e..9a07dc2f8 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -76,9 +76,9 @@ class ResponseTypesTest(unittest.TestCase): mappings = [ ({'url': 'http://www.example.com/data.csv'}, TextResponse), ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), - 'url': 'http://www.example.com/item/'}, HtmlResponse), # Failing with xtractmime, returning TextResponse expected HtmlResponse + 'url': 'http://www.example.com/item/'}, HtmlResponse), ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), - 'url': 'http://www.example.com/page/'}, Response), # Failing with xtractmime, returning TextResponse expected Response + 'url': 'http://www.example.com/page/'}, Response), ] for source, cls in mappings: retcls = responsetypes.from_args(**source) @@ -86,19 +86,19 @@ class ResponseTypesTest(unittest.TestCase): def test_from_args_post_xtractmime(self): mappings = [ - ({'body': b'Some plain text data with tabs and null bytes'}, TextResponse), - ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, + ({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse), + ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), # different behaviour with http and non-http urls - ({'body': b'\x00\xfe\xff', 'url': 'http://www.example.com/item/', - 'headers': Headers({'Content-Type': b'text/plain'})}, Response), - ({'body': b'\x00\xfe\xff', 'url': '://www.example.com/item/', + ({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/', + 'headers': Headers({'Content-Type': b'text/plain'})}, Response), + ({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/', 'headers': Headers({'Content-Type': b'text/plain'})}, TextResponse), - ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), # Failing with xtractmime, return TextResponse expected HtmlResponse - ({'body': b'Some plain text\ndata with tabs\t and null bytes\0'}, Response), # earlier expected to be binary response, refer "test_from_body()" + ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), + ({'body': b'Some plain text data\1 with tabs and\n null bytes\0'}, Response), ({'body': b'Hello'}, HtmlResponse), ({'body': b' Date: Sun, 1 Aug 2021 17:17:20 +0530 Subject: [PATCH 14/81] Refactor response class --- scrapy/responsetypes.py | 25 +++++++++++++++++++++++-- 1 file changed, 23 insertions(+), 2 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index fd7310c80..f22409643 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -10,9 +10,15 @@ from warnings import warn from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime from xtractmime._utils import contains_binary +from xtractmime.mimegroups import ( + is_html_mime_type, + is_javascript_mime_type, + is_json_mime_type, + is_xml_mime_type, +) from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.http import Response +from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse from scrapy.utils.misc import load_object from scrapy.utils.python import binary_is_text, to_bytes, to_unicode @@ -136,6 +142,21 @@ class ResponseTypes: return None + def _guess_response_type(self, mime_type): + if not mime_type: + return Response + if is_html_mime_type(mime_type): + return HtmlResponse + if is_xml_mime_type(mime_type): + return XmlResponse + if ( + mime_type.startswith(b'text/') + or is_json_mime_type(mime_type) + or is_javascript_mime_type(mime_type) + ): + return TextResponse + return Response + def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on the given arguments.""" @@ -156,7 +177,7 @@ class ResponseTypes: http_origin = not url or urlparse(url).scheme in ("http", "https") content_types = self._guess_content_type(headers=headers, url=url, filename=filename) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) - cls = self.from_mimetype(mime_type.decode()) + cls = self._guess_response_type(mime_type) return cls From 841a218431a7739d079fe54cea727954594093c5 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Mon, 2 Aug 2021 14:26:54 +0530 Subject: [PATCH 15/81] Small fix --- scrapy/responsetypes.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index f22409643..7116636ca 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -163,15 +163,15 @@ class ResponseTypes: if not body: body = b'' + body = body[:RESOURCE_HEADER_BUFFER_LENGTH] contains_binary_bytes = False - for index in range(len(body)): if body[index:index + 1] != b"\x00" and contains_binary(body[index:index + 1]): contains_binary_bytes = True break if not contains_binary_bytes: - body = body[:RESOURCE_HEADER_BUFFER_LENGTH].replace(b"\x00", b"") + body = body.replace(b"\x00", b"") cls = Response http_origin = not url or urlparse(url).scheme in ("http", "https") From 0b52719e5f12f4241d834c9f3ddc9965ef9678c5 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sun, 8 Aug 2021 14:19:19 +0530 Subject: [PATCH 16/81] Add is_binary_data() --- scrapy/responsetypes.py | 5 ++--- tox.ini | 2 +- 2 files changed, 3 insertions(+), 4 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 7116636ca..147fd2628 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -8,8 +8,7 @@ from io import StringIO from urllib.parse import urlparse from warnings import warn -from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime -from xtractmime._utils import contains_binary +from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime, is_binary_data from xtractmime.mimegroups import ( is_html_mime_type, is_javascript_mime_type, @@ -166,7 +165,7 @@ class ResponseTypes: body = body[:RESOURCE_HEADER_BUFFER_LENGTH] contains_binary_bytes = False for index in range(len(body)): - if body[index:index + 1] != b"\x00" and contains_binary(body[index:index + 1]): + if body[index:index + 1] != b"\x00" and is_binary_data(body[index:index + 1]): contains_binary_bytes = True break diff --git a/tox.ini b/tox.ini index 58e902ec1..86e1d78f8 100644 --- a/tox.ini +++ b/tox.ini @@ -17,7 +17,7 @@ deps = #mitmproxy >= 5.3.0; python_version >= '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4; python_version >= '3.7' and python_version < '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4, < 5; python_version >= '3.6' and python_version < '3.7' and platform_system != 'Windows' and implementation_name != 'pypy' - git+https://github.com/scrapy/xtractmime.git@html-fix#egg=xtractmime + git+https://github.com/scrapy/xtractmime.git@binary#egg=xtractmime # Extras botocore>=1.4.87 passenv = From 2ba2f49ac52da715b9499d05ab092552b435024a Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sun, 8 Aug 2021 19:35:48 +0530 Subject: [PATCH 17/81] Add more tests --- tests/test_responsetypes.py | 25 +++++++++++++++---------- 1 file changed, 15 insertions(+), 10 deletions(-) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 9a07dc2f8..52b0e2507 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -79,6 +79,17 @@ class ResponseTypesTest(unittest.TestCase): 'url': 'http://www.example.com/item/'}, HtmlResponse), ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), 'url': 'http://www.example.com/page/'}, Response), + ({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse), + ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, + Response), + ({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/', + 'headers': Headers({'Content-Type': b'text/plain'})}, TextResponse), + ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), + ({'body': b'Hello'}, HtmlResponse), + ({'body': b'Hello'}, HtmlResponse), - ({'body': b' Date: Mon, 9 Aug 2021 17:28:12 +0530 Subject: [PATCH 18/81] ADD xtractmime to pinnedtests --- scrapy/responsetypes.py | 2 +- tests/test_responsetypes.py | 1 + tox.ini | 3 ++- 3 files changed, 4 insertions(+), 2 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 147fd2628..e60ca04dd 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -136,7 +136,7 @@ class ResponseTypes: mimetype, encoding = self.mimetypes.guess_type(filename) if encoding: return (f"application/{encoding}".encode(),) - else: + elif mimetype: return (mimetype.encode(),) return None diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 52b0e2507..2d3ae93ce 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -90,6 +90,7 @@ class ResponseTypesTest(unittest.TestCase): ({'filename': 'file.pdf'}, Response), ({'url': 'http://www.example.com/item/file.pdf'}, Response), ({'body': b'Some plain text data\1\2 with tabs and\n null bytes\0'}, Response), + ({'filename': '/tmp/temp^'}, TextResponse), ] for source, cls in mappings: retcls = responsetypes.from_args(**source) diff --git a/tox.ini b/tox.ini index 86e1d78f8..d31ea5517 100644 --- a/tox.ini +++ b/tox.ini @@ -17,7 +17,7 @@ deps = #mitmproxy >= 5.3.0; python_version >= '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4; python_version >= '3.7' and python_version < '3.9' and implementation_name != 'pypy' mitmproxy >= 4.0.4, < 5; python_version >= '3.6' and python_version < '3.7' and platform_system != 'Windows' and implementation_name != 'pypy' - git+https://github.com/scrapy/xtractmime.git@binary#egg=xtractmime + git+https://github.com/scrapy/xtractmime.git@main#egg=xtractmime # Extras botocore>=1.4.87 passenv = @@ -79,6 +79,7 @@ deps = Twisted[http2]==17.9.0 w3lib==1.17.0 zope.interface==4.1.3 + git+https://github.com/scrapy/xtractmime.git@main#egg=xtractmime -rtests/requirements-py3.txt # mitmproxy 4.0.4+ requires upgrading some of the pinned dependencies From 4169fd184f478653139cf37aedc7e96fe8d38bde Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Wed, 11 Aug 2021 17:14:38 +0530 Subject: [PATCH 19/81] More tests --- scrapy/responsetypes.py | 37 +++++++++++++++++++++++-------------- tests/test_responsetypes.py | 26 +++++++++++++++++++------- 2 files changed, 42 insertions(+), 21 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index e60ca04dd..cbb1538bc 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -149,36 +149,45 @@ class ResponseTypes: if is_xml_mime_type(mime_type): return XmlResponse if ( - mime_type.startswith(b'text/') + mime_type.startswith(b"text/") or is_json_mime_type(mime_type) or is_javascript_mime_type(mime_type) + or mime_type + in ( + b"application/x-json", + b"application/json-amazonui-streaming", + b"application/x-javascript", + ) ): return TextResponse return Response - def from_args(self, headers=None, url=None, filename=None, body=None): - """Guess the most appropriate Response class based on - the given arguments.""" - if not body: - body = b'' + def _remove_nul_byte_from_text(self, text): + """Return the text with removed null byte (b"\x00") if there are no other + binary bytes in the text, otherwise return the text as-is. - body = body[:RESOURCE_HEADER_BUFFER_LENGTH] + Based on https://github.com/scrapy/scrapy/issues/2481""" contains_binary_bytes = False - for index in range(len(body)): - if body[index:index + 1] != b"\x00" and is_binary_data(body[index:index + 1]): + + for index in range(len(text)): + if text[index:index + 1] != b"\x00" and is_binary_data(text[index:index + 1]): contains_binary_bytes = True break if not contains_binary_bytes: - body = body.replace(b"\x00", b"") + text = text.replace(b"\x00", b"") - cls = Response + return text + + def from_args(self, headers=None, url=None, filename=None, body=None): + """Guess the most appropriate Response class based on + the given arguments.""" + body = body or b'' + body = self._remove_nul_byte_from_text(body[:RESOURCE_HEADER_BUFFER_LENGTH]) http_origin = not url or urlparse(url).scheme in ("http", "https") content_types = self._guess_content_type(headers=headers, url=url, filename=filename) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) - cls = self._guess_response_type(mime_type) - - return cls + return self._guess_response_type(mime_type) responsetypes = ResponseTypes() diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 2d3ae93ce..12fe6d66f 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -83,14 +83,14 @@ class ResponseTypesTest(unittest.TestCase): ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, Response), ({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/', - 'headers': Headers({'Content-Type': b'text/plain'})}, TextResponse), + 'headers': Headers({'Content-Type': ['text/plain']})}, TextResponse), ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), ({'body': b'Hello'}, HtmlResponse), ({'body': b''}, TextResponse), + ({'body': b'this is not Date: Fri, 13 Aug 2021 01:05:37 +0530 Subject: [PATCH 20/81] Change _guess_content_type --- scrapy/responsetypes.py | 56 ++++++++++++++++++------------------- tests/test_responsetypes.py | 20 +++++++++---- 2 files changed, 42 insertions(+), 34 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index cbb1538bc..998c25569 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -123,24 +123,6 @@ class ResponseTypes: else: return self.from_mimetype('text') - def _guess_content_type(self, headers=None, url=None, filename=None): - if headers and b'Content-Type' in headers: - return tuple(headers.getlist(b'Content-Type')) - - if headers and b'Content-Disposition' in headers: - filename = headers.get(b'Content-Disposition').split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() - elif url: - filename = url - - if filename: - mimetype, encoding = self.mimetypes.guess_type(filename) - if encoding: - return (f"application/{encoding}".encode(),) - elif mimetype: - return (mimetype.encode(),) - - return None - def _guess_response_type(self, mime_type): if not mime_type: return Response @@ -152,8 +134,7 @@ class ResponseTypes: mime_type.startswith(b"text/") or is_json_mime_type(mime_type) or is_javascript_mime_type(mime_type) - or mime_type - in ( + or mime_type in ( b"application/x-json", b"application/json-amazonui-streaming", b"application/x-javascript", @@ -162,22 +143,39 @@ class ResponseTypes: return TextResponse return Response + def _guess_content_type(self, body=None, headers=None, url=None, filename=None): + mimetype = None + + if headers and b'Content-Type' in headers: + mimetype = tuple(headers.getlist(b'Content-Type')) + else: + if headers and b'Content-Disposition' in headers: + filename = headers.get(b'Content-Disposition').split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() + elif url: + filename = url + + if filename: + mimetype, encoding = self.mimetypes.guess_type(filename) + if encoding: + mimetype = (f"application/{encoding}".encode(),) + elif mimetype: + mimetype = (mimetype.encode(),) + + if mimetype and self._guess_response_type(mimetype[-1]) is Response: + return None + + return mimetype + def _remove_nul_byte_from_text(self, text): """Return the text with removed null byte (b"\x00") if there are no other binary bytes in the text, otherwise return the text as-is. Based on https://github.com/scrapy/scrapy/issues/2481""" - contains_binary_bytes = False - for index in range(len(text)): if text[index:index + 1] != b"\x00" and is_binary_data(text[index:index + 1]): - contains_binary_bytes = True - break + return text - if not contains_binary_bytes: - text = text.replace(b"\x00", b"") - - return text + return text.replace(b"\x00", b"") def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on @@ -185,7 +183,7 @@ class ResponseTypes: body = body or b'' body = self._remove_nul_byte_from_text(body[:RESOURCE_HEADER_BUFFER_LENGTH]) http_origin = not url or urlparse(url).scheme in ("http", "https") - content_types = self._guess_content_type(headers=headers, url=url, filename=filename) + content_types = self._guess_content_type(body=body, headers=headers, url=url, filename=filename) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) return self._guess_response_type(mime_type) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 12fe6d66f..c62460c84 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -73,11 +73,15 @@ class ResponseTypesTest(unittest.TestCase): assert retcls is cls, f"{source} ==> {retcls} != {cls}" def test_from_args_pre_xtractmime(self): + """Each of the following test cases remains unaffected after + using xtractmime for MIME sniffing""" mappings = [ ({'url': 'http://www.example.com/data.csv'}, TextResponse), ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), 'url': 'http://www.example.com/item/'}, HtmlResponse), ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), + 'url': 'http://www.example.com/page/'}, TextResponse), + ({'body': b'\x01\x02', 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), 'url': 'http://www.example.com/page/'}, Response), ({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse), ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, @@ -87,25 +91,31 @@ class ResponseTypesTest(unittest.TestCase): ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), ({'body': b'Hello'}, HtmlResponse), ({'body': b' {retcls} != {cls}" def test_from_args_post_xtractmime(self): + """Each of the following test cases got affected after + using xtractmime for MIME sniffing""" mappings = [ ({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/', 'headers': Headers({'Content-Type': ['text/plain']})}, Response), ({'filename': '/tmp/temp^'}, TextResponse), ({'body': b'%PDF-1.4'}, Response), - ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), + ({'headers': Headers({'Content-Type': ['application/pdf']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/ecmascript']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/ld+json']})}, TextResponse), - ({'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse), ({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})}, HtmlResponse), ({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']})}, @@ -113,7 +123,7 @@ class ResponseTypesTest(unittest.TestCase): ({'body': b'Non HTML', 'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})}, HtmlResponse), ({'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, - Response), + TextResponse), ({'body': b'\x0c\x1b'}, TextResponse), ({'body': b'this is not '}, TextResponse), ({'body': b'this is not Date: Fri, 13 Aug 2021 01:15:05 +0530 Subject: [PATCH 21/81] Small fix --- tox.ini | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tox.ini b/tox.ini index 8e666f151..3bbc1c05d 100644 --- a/tox.ini +++ b/tox.ini @@ -84,7 +84,7 @@ deps = w3lib==1.17.0 zope.interface==4.1.3 git+https://github.com/scrapy/xtractmime.git@main#egg=xtractmime - -rtests/requirements-py3.txt + -rtests/requirements.txt # mitmproxy 4.0.4+ requires upgrading some of the pinned dependencies # above, hence we do not install it in pinned environments at the moment From c81605abe62e1ab6bcd8e46e425d0ede3f785c20 Mon Sep 17 00:00:00 2001 From: Akshay Sharma Date: Sun, 29 Aug 2021 13:55:02 -0400 Subject: [PATCH 22/81] Move tests to post_xtractmime --- tests/test_responsetypes.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index c62460c84..15a7ea680 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -79,8 +79,6 @@ class ResponseTypesTest(unittest.TestCase): ({'url': 'http://www.example.com/data.csv'}, TextResponse), ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), 'url': 'http://www.example.com/item/'}, HtmlResponse), - ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), - 'url': 'http://www.example.com/page/'}, TextResponse), ({'body': b'\x01\x02', 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), 'url': 'http://www.example.com/page/'}, Response), ({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse), @@ -91,10 +89,6 @@ class ResponseTypesTest(unittest.TestCase): ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), ({'body': b'Hello'}, HtmlResponse), ({'body': b''}, TextResponse), ({'body': b'this is not Date: Mon, 20 Sep 2021 22:44:13 -0400 Subject: [PATCH 23/81] Add more tests --- tests/test_responsetypes.py | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 15a7ea680..16ce36233 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -120,8 +120,15 @@ class ResponseTypesTest(unittest.TestCase): 'Content-Type': ['text/html']})}, HtmlResponse), ({'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), - ({'url': b'http://www.example.com/page/file.html', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, - TextResponse), + ({'url': b'http://www.example.com/page/file.html', + 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), + ({'url': 'http://www.example.com/item/file.xml', + 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), + ({'url': 'http://www.example.com/item/file.html', + 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), + ({'url': 'http://www.example.com/item/file.xml', + 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"'], + 'Content-Type': 'application/octet-stream'})}, TextResponse), ({'body': b'\x0c\x1b'}, TextResponse), ({'body': b'this is not '}, TextResponse), ({'body': b'this is not Date: Mon, 20 Jun 2022 20:15:33 -0400 Subject: [PATCH 24/81] Refactored _guess_content_type --- scrapy/responsetypes.py | 89 ++++++++++++++++++++++++++++--------- tests/test_responsetypes.py | 30 ++++++------- 2 files changed, 82 insertions(+), 37 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 998c25569..816b1f040 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -3,8 +3,10 @@ This module implements a class which returns the appropriate Response class based on different criteria. """ from mimetypes import MimeTypes +from operator import truediv from pkgutil import get_data from io import StringIO +from traceback import format_exception_only from urllib.parse import urlparse from warnings import warn @@ -48,6 +50,12 @@ class ResponseTypes: self.mimetypes.readfp(StringIO(mimedata)) for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) + + self.prioritized_mime_type_checkers = ( + is_html_mime_type, + is_xml_mime_type, + self._is_text_mime_type, + ) def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" @@ -122,6 +130,20 @@ class ResponseTypes: return self.from_mimetype('text/xml') else: return self.from_mimetype('text') + + def _is_text_mime_type(self, mime_type): + if ( + mime_type.startswith(b"text/") + or is_json_mime_type(mime_type) + or is_javascript_mime_type(mime_type) + or mime_type in ( + b"application/x-json", + b"application/json-amazonui-streaming", + b"application/x-javascript", + ) + ): + return True + return False def _guess_response_type(self, mime_type): if not mime_type: @@ -142,30 +164,52 @@ class ResponseTypes: ): return TextResponse return Response - - def _guess_content_type(self, body=None, headers=None, url=None, filename=None): - mimetype = None - - if headers and b'Content-Type' in headers: - mimetype = tuple(headers.getlist(b'Content-Type')) - else: - if headers and b'Content-Disposition' in headers: - filename = headers.get(b'Content-Disposition').split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() - elif url: - filename = url - - if filename: - mimetype, encoding = self.mimetypes.guess_type(filename) - if encoding: - mimetype = (f"application/{encoding}".encode(),) - elif mimetype: - mimetype = (mimetype.encode(),) - - if mimetype and self._guess_response_type(mimetype[-1]) is Response: - return None + + def _guess_type(self, filename=None): + mimetype, encoding = self.mimetypes.guess_type(filename) + if encoding: + mimetype = f"application/{encoding}".encode() + elif mimetype: + mimetype = mimetype.encode() return mimetype + def _guess_content_type(self, body=None, headers=None, url=None, filename=None): + if headers and b'Content-Type' in headers: + content_type_mime_type = headers.getlist(b'Content-Type')[-1] + else: + content_type_mime_type = None + + if headers and b'Content-Disposition' in headers: + content_disposition_mime_type = headers.get(b'Content-Disposition').split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() + content_disposition_mime_type = self._guess_type(content_disposition_mime_type) + else: + content_disposition_mime_type = None + + url_mime_type = self._guess_type(url) if url else None + filename_mime_type = self._guess_type(filename) if filename else None + + candidate_mime_types = ( + content_type_mime_type, + content_disposition_mime_type, + url_mime_type, + filename_mime_type, + ) + + for mime_type_checker in self.prioritized_mime_type_checkers: + for candidate_mime_type in candidate_mime_types: + if ( + candidate_mime_type is not None + and mime_type_checker(candidate_mime_type) + ): + return candidate_mime_type + return ( + content_type_mime_type + or content_disposition_mime_type + or url_mime_type + or filename_mime_type + ) + def _remove_nul_byte_from_text(self, text): """Return the text with removed null byte (b"\x00") if there are no other binary bytes in the text, otherwise return the text as-is. @@ -183,7 +227,8 @@ class ResponseTypes: body = body or b'' body = self._remove_nul_byte_from_text(body[:RESOURCE_HEADER_BUFFER_LENGTH]) http_origin = not url or urlparse(url).scheme in ("http", "https") - content_types = self._guess_content_type(body=body, headers=headers, url=url, filename=filename) + content_types = (self._guess_content_type(body=body, headers=headers, url=url, filename=filename),) + content_types = None if content_types == (None,) else content_types mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) return self._guess_response_type(mime_type) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 16ce36233..c309b75f0 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -94,6 +94,20 @@ class ResponseTypesTest(unittest.TestCase): ({'headers': Headers({'Content-Type': ['application/x-json']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/json-amazonui-streaming']})}, TextResponse), + ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), + 'url': 'http://www.example.com/page/'}, Response), + ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), + ({'url': 'http://www.example.com/page/file.html', + 'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse), + ({'url': 'http://www.example.com/item/file.xml', + 'headers': Headers({'Content-Type': 'application/octet-stream'})}, XmlResponse), + ({'url': 'http://www.example.com/item/file.html', + 'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse), + ({'url': 'http://www.example.com/item/file.xml', + 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"'], + 'Content-Type': 'application/octet-stream'})}, XmlResponse), + ({'url': 'http://www.example.com/item/file.pdf'}, Response), + ({'filename': 'file.pdf'}, Response), ] for source, cls in mappings: retcls = responsetypes.from_args(**source) @@ -103,13 +117,10 @@ class ResponseTypesTest(unittest.TestCase): """Each of the following test cases got affected after using xtractmime for MIME sniffing""" mappings = [ - ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), - 'url': 'http://www.example.com/page/'}, TextResponse), ({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/', 'headers': Headers({'Content-Type': ['text/plain']})}, Response), ({'filename': '/tmp/temp^'}, TextResponse), ({'body': b'%PDF-1.4'}, Response), - ({'headers': Headers({'Content-Type': ['application/pdf']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/ecmascript']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/ld+json']})}, TextResponse), ({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})}, @@ -119,21 +130,10 @@ class ResponseTypesTest(unittest.TestCase): ({'body': b'Non HTML', 'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})}, HtmlResponse), ({'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, - TextResponse), - ({'url': b'http://www.example.com/page/file.html', - 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), - ({'url': 'http://www.example.com/item/file.xml', - 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), - ({'url': 'http://www.example.com/item/file.html', - 'headers': Headers({'Content-Type': 'application/octet-stream'})}, TextResponse), - ({'url': 'http://www.example.com/item/file.xml', - 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"'], - 'Content-Type': 'application/octet-stream'})}, TextResponse), + Response), ({'body': b'\x0c\x1b'}, TextResponse), ({'body': b'this is not '}, TextResponse), ({'body': b'this is not Date: Mon, 20 Jun 2022 21:10:06 -0400 Subject: [PATCH 25/81] fix static checks --- scrapy/responsetypes.py | 30 ++++++++++++++++++------------ tests/test_responsetypes.py | 4 +++- 2 files changed, 21 insertions(+), 13 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 816b1f040..4140d9de0 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -3,10 +3,8 @@ This module implements a class which returns the appropriate Response class based on different criteria. """ from mimetypes import MimeTypes -from operator import truediv from pkgutil import get_data from io import StringIO -from traceback import format_exception_only from urllib.parse import urlparse from warnings import warn @@ -50,7 +48,7 @@ class ResponseTypes: self.mimetypes.readfp(StringIO(mimedata)) for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) - + self.prioritized_mime_type_checkers = ( is_html_mime_type, is_xml_mime_type, @@ -124,13 +122,15 @@ class ResponseTypes: chunk = to_bytes(chunk) if not binary_is_text(chunk): return self.from_mimetype('application/octet-stream') - elif b"" in chunk.lower(): + lowercase_chunk = chunk.lower() + if b"" in lowercase_chunk: return self.from_mimetype('text/html') - elif b"' in lowercase_chunk: + return self.from_mimetype('text/html') + return self.from_mimetype('text') + def _is_text_mime_type(self, mime_type): if ( mime_type.startswith(b"text/") @@ -164,7 +164,7 @@ class ResponseTypes: ): return TextResponse return Response - + def _guess_type(self, filename=None): mimetype, encoding = self.mimetypes.guess_type(filename) if encoding: @@ -181,11 +181,17 @@ class ResponseTypes: content_type_mime_type = None if headers and b'Content-Disposition' in headers: - content_disposition_mime_type = headers.get(b'Content-Disposition').split(b';')[-1].split(b'=')[-1].strip(b'"\'').decode() + content_disposition_mime_type = ( + headers.get(b"Content-Disposition") + .split(b";")[-1] + .split(b"=")[-1] + .strip(b"\"'") + .decode() + ) content_disposition_mime_type = self._guess_type(content_disposition_mime_type) else: content_disposition_mime_type = None - + url_mime_type = self._guess_type(url) if url else None filename_mime_type = self._guess_type(filename) if filename else None @@ -199,7 +205,7 @@ class ResponseTypes: for mime_type_checker in self.prioritized_mime_type_checkers: for candidate_mime_type in candidate_mime_types: if ( - candidate_mime_type is not None + candidate_mime_type is not None and mime_type_checker(candidate_mime_type) ): return candidate_mime_type diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index c309b75f0..e891a1c24 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -90,6 +90,8 @@ class ResponseTypesTest(unittest.TestCase): ({'body': b'Hello'}, HtmlResponse), ({'body': b'\n.'}, HtmlResponse), ({'body': b'\x01\x02', 'headers': Headers({'Content-Type': ['application/pdf']})}, Response), ({'headers': Headers({'Content-Type': ['application/x-json']})}, TextResponse), ({'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse), @@ -97,7 +99,7 @@ class ResponseTypesTest(unittest.TestCase): ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), 'url': 'http://www.example.com/page/'}, Response), ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), - ({'url': 'http://www.example.com/page/file.html', + ({'url': 'http://www.example.com/page/file.html', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse), ({'url': 'http://www.example.com/item/file.xml', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, XmlResponse), From ef6290ef1b955cbe06a47053eda4a8064521713c Mon Sep 17 00:00:00 2001 From: Akshay Sharma <42249933+akshaysharmajs@users.noreply.github.com> Date: Tue, 21 Jun 2022 02:44:18 -0400 Subject: [PATCH 26/81] url ending fix --- scrapy/extensions/httpcache.py | 1 + scrapy/responsetypes.py | 1 + tests/test_downloadermiddleware_httpcache.py | 3 ++- 3 files changed, 4 insertions(+), 1 deletion(-) diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 843e14812..7733ffa5c 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -241,6 +241,7 @@ class DbmCacheStorage: headers = Headers(data['headers']) body = data['body'] respcls = responsetypes.from_args(headers=headers, url=url, body=body) + print(respcls) response = respcls(url=url, headers=headers, status=status, body=body) return response diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 4140d9de0..95252498b 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -235,6 +235,7 @@ class ResponseTypes: http_origin = not url or urlparse(url).scheme in ("http", "https") content_types = (self._guess_content_type(body=body, headers=headers, url=url, filename=filename),) content_types = None if content_types == (None,) else content_types + print(content_types) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) return self._guess_response_type(mime_type) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 928c007f5..c84c2bdeb 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -128,12 +128,13 @@ class DefaultStorageTest(_BaseTest): with self._storage() as storage: assert storage.retrieve_response(self.spider, self.request) is None response = Response( - 'http://www.example.com', + 'http://www.example.com/', body=b'\n.', status=202, ) storage.store_response(self.spider, self.request, response) cached_response = storage.retrieve_response(self.spider, self.request) + print(cached_response) self.assertIsInstance(cached_response, HtmlResponse) self.assertEqualResponse(response, cached_response) From 8340394bcc2c1618da69e1b8b2f1a9ead20601c2 Mon Sep 17 00:00:00 2001 From: Akshay Sharma <42249933+akshaysharmajs@users.noreply.github.com> Date: Tue, 21 Jun 2022 02:45:39 -0400 Subject: [PATCH 27/81] removed prints --- scrapy/extensions/httpcache.py | 1 - scrapy/responsetypes.py | 1 - tests/test_downloadermiddleware_httpcache.py | 1 - 3 files changed, 3 deletions(-) diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 7733ffa5c..843e14812 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -241,7 +241,6 @@ class DbmCacheStorage: headers = Headers(data['headers']) body = data['body'] respcls = responsetypes.from_args(headers=headers, url=url, body=body) - print(respcls) response = respcls(url=url, headers=headers, status=status, body=body) return response diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 95252498b..4140d9de0 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -235,7 +235,6 @@ class ResponseTypes: http_origin = not url or urlparse(url).scheme in ("http", "https") content_types = (self._guess_content_type(body=body, headers=headers, url=url, filename=filename),) content_types = None if content_types == (None,) else content_types - print(content_types) mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) return self._guess_response_type(mime_type) diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index c84c2bdeb..7945d2a61 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -134,7 +134,6 @@ class DefaultStorageTest(_BaseTest): ) storage.store_response(self.spider, self.request, response) cached_response = storage.retrieve_response(self.spider, self.request) - print(cached_response) self.assertIsInstance(cached_response, HtmlResponse) self.assertEqualResponse(response, cached_response) From 9044ceb824be1f58b6ea41d8436a52c725b6d2e0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 21 Jun 2022 10:27:12 +0200 Subject: [PATCH 28/81] Refactor responsetypes --- scrapy/responsetypes.py | 236 +++++++++++++++++++----------------- tests/test_responsetypes.py | 41 +++++-- 2 files changed, 158 insertions(+), 119 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 4140d9de0..77b190a98 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -22,6 +22,114 @@ from scrapy.utils.misc import load_object from scrapy.utils.python import binary_is_text, to_bytes, to_unicode +_CONTENT_ENCODING_MAP = { + 'br': b'application/brotli', + 'deflate': b'application/zip', + 'gzip': b'application/gzip', +} +_MIMETYPES = MimeTypes() +_MIMETYPES.readfp(StringIO(get_data('scrapy', 'mime.types').decode())) + + +def _is_other_text_mime_type(mime_type): + value = ( + mime_type.startswith(b'text/') + or is_json_mime_type(mime_type) + or is_javascript_mime_type(mime_type) + or mime_type in ( + b'application/x-json', + b'application/json-amazonui-streaming', + b'application/x-javascript', + ) + ) + return value + + +_PRIORITIZED_MIME_TYPE_CHECKERS = ( + is_html_mime_type, + is_xml_mime_type, + _is_other_text_mime_type, +) + + +def _content_type_from_metadata(*, headers=None, url_path=None, filename=None): + if headers and b'Content-Type' in headers: + content_type_mime_type = headers.getlist(b'Content-Type')[-1] + else: + content_type_mime_type = None + + if headers and b'Content-Disposition' in headers: + _filename = ( + headers.get(b"Content-Disposition") + .split(b";")[-1] + .split(b"=")[-1] + .strip(b"\"'") + .decode() + ) + content_disposition_mime_type = _mime_type_from_path(_filename) + else: + content_disposition_mime_type = None + + filename_mime_type = _mime_type_from_path(filename) if filename else None + url_mime_type = _mime_type_from_path(url_path) if url_path else None + + candidate_mime_types = tuple( + mime_type + for mime_type in ( + content_type_mime_type, + content_disposition_mime_type, + filename_mime_type, + url_mime_type, + ) + if mime_type is not None + ) + for mime_type_checker in _PRIORITIZED_MIME_TYPE_CHECKERS: + for candidate_mime_type in candidate_mime_types: + if mime_type_checker(candidate_mime_type): + return candidate_mime_type + return ( + content_type_mime_type + or content_disposition_mime_type + or filename_mime_type + or url_mime_type + ) + +def _mime_type_from_path(path): + mimetype, encoding = _MIMETYPES.guess_type(path, strict=False) + encoding_mime_type = _CONTENT_ENCODING_MAP.get(encoding, None) + if encoding_mime_type: + return encoding_mime_type + if mimetype: + return mimetype.encode() + return None + +def _remove_nul_byte_from_text(text): + """Return the text with removed null byte (b'\x00') if there are no other + binary bytes in the text, otherwise return the text as-is. + + Based on https://github.com/scrapy/scrapy/issues/2481 + """ + for index in range(len(text)): + if ( + text[index:index + 1] != b'\x00' + and is_binary_data(text[index:index + 1]) + ): + return text + + return text.replace(b'\x00', b'') + +def _response_type_from_mime_type(mime_type): + if not mime_type: + return Response + if is_html_mime_type(mime_type): + return HtmlResponse + if is_xml_mime_type(mime_type): + return XmlResponse + if _is_other_text_mime_type(mime_type): + return TextResponse + return Response + + class ResponseTypes: CLASSES = { @@ -43,18 +151,9 @@ class ResponseTypes: def __init__(self): self.classes = {} - self.mimetypes = MimeTypes() - mimedata = get_data('scrapy', 'mime.types').decode('utf8') - self.mimetypes.readfp(StringIO(mimedata)) for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) - self.prioritized_mime_type_checkers = ( - is_html_mime_type, - is_xml_mime_type, - self._is_text_mime_type, - ) - def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" if mimetype is None: @@ -105,7 +204,7 @@ class ResponseTypes: """Return the most appropriate Response class from a file name""" warn('ResponseTypes.from_filename is deprecated, ' 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) - mimetype, encoding = self.mimetypes.guess_type(filename) + mimetype, encoding = _MIMETYPES.guess_type(filename) if mimetype and not encoding: return self.from_mimetype(mimetype) else: @@ -131,112 +230,25 @@ class ResponseTypes: return self.from_mimetype('text/html') return self.from_mimetype('text') - def _is_text_mime_type(self, mime_type): - if ( - mime_type.startswith(b"text/") - or is_json_mime_type(mime_type) - or is_javascript_mime_type(mime_type) - or mime_type in ( - b"application/x-json", - b"application/json-amazonui-streaming", - b"application/x-javascript", - ) - ): - return True - return False - - def _guess_response_type(self, mime_type): - if not mime_type: - return Response - if is_html_mime_type(mime_type): - return HtmlResponse - if is_xml_mime_type(mime_type): - return XmlResponse - if ( - mime_type.startswith(b"text/") - or is_json_mime_type(mime_type) - or is_javascript_mime_type(mime_type) - or mime_type in ( - b"application/x-json", - b"application/json-amazonui-streaming", - b"application/x-javascript", - ) - ): - return TextResponse - return Response - - def _guess_type(self, filename=None): - mimetype, encoding = self.mimetypes.guess_type(filename) - if encoding: - mimetype = f"application/{encoding}".encode() - elif mimetype: - mimetype = mimetype.encode() - - return mimetype - - def _guess_content_type(self, body=None, headers=None, url=None, filename=None): - if headers and b'Content-Type' in headers: - content_type_mime_type = headers.getlist(b'Content-Type')[-1] - else: - content_type_mime_type = None - - if headers and b'Content-Disposition' in headers: - content_disposition_mime_type = ( - headers.get(b"Content-Disposition") - .split(b";")[-1] - .split(b"=")[-1] - .strip(b"\"'") - .decode() - ) - content_disposition_mime_type = self._guess_type(content_disposition_mime_type) - else: - content_disposition_mime_type = None - - url_mime_type = self._guess_type(url) if url else None - filename_mime_type = self._guess_type(filename) if filename else None - - candidate_mime_types = ( - content_type_mime_type, - content_disposition_mime_type, - url_mime_type, - filename_mime_type, - ) - - for mime_type_checker in self.prioritized_mime_type_checkers: - for candidate_mime_type in candidate_mime_types: - if ( - candidate_mime_type is not None - and mime_type_checker(candidate_mime_type) - ): - return candidate_mime_type - return ( - content_type_mime_type - or content_disposition_mime_type - or url_mime_type - or filename_mime_type - ) - - def _remove_nul_byte_from_text(self, text): - """Return the text with removed null byte (b"\x00") if there are no other - binary bytes in the text, otherwise return the text as-is. - - Based on https://github.com/scrapy/scrapy/issues/2481""" - for index in range(len(text)): - if text[index:index + 1] != b"\x00" and is_binary_data(text[index:index + 1]): - return text - - return text.replace(b"\x00", b"") - def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on the given arguments.""" body = body or b'' - body = self._remove_nul_byte_from_text(body[:RESOURCE_HEADER_BUFFER_LENGTH]) - http_origin = not url or urlparse(url).scheme in ("http", "https") - content_types = (self._guess_content_type(body=body, headers=headers, url=url, filename=filename),) - content_types = None if content_types == (None,) else content_types - mime_type = extract_mime(body, content_types=content_types, http_origin=http_origin) - return self._guess_response_type(mime_type) + body = _remove_nul_byte_from_text(body[:RESOURCE_HEADER_BUFFER_LENGTH]) + url_parts = urlparse(url) if url else url + content_type = _content_type_from_metadata( + headers=headers, + url_path=url_parts.path if url_parts else url_parts, + filename=filename, + ) + content_types = (content_type,) if content_type else None + http_origin = not url or url_parts.scheme in ("http", "https") + mime_type = extract_mime( + body, + content_types=content_types, + http_origin=http_origin, + ) + return _response_type_from_mime_type(mime_type) responsetypes = ResponseTypes() diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index e891a1c24..3263dcb24 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -1,9 +1,26 @@ import unittest -from scrapy.responsetypes import responsetypes +from scrapy.responsetypes import _MIMETYPES, responsetypes, ResponseTypes from scrapy.http import Response, TextResponse, XmlResponse, HtmlResponse, Headers +class PreXtractmimeResponseTypes(ResponseTypes): + def from_args(self, headers=None, url=None, filename=None, body=None): + cls = Response + if headers is not None: + cls = self.from_headers(headers) + if cls is Response and url is not None: + cls = self.from_filename(url) + if cls is Response and filename is not None: + cls = self.from_filename(filename) + if cls is Response and body is not None: + cls = self.from_body(body) + return cls + + +_PRE_XTRACTMIME_RESPONSE_TYPES = PreXtractmimeResponseTypes() + + class ResponseTypesTest(unittest.TestCase): def test_from_filename(self): @@ -110,10 +127,14 @@ class ResponseTypesTest(unittest.TestCase): 'Content-Type': 'application/octet-stream'})}, XmlResponse), ({'url': 'http://www.example.com/item/file.pdf'}, Response), ({'filename': 'file.pdf'}, Response), + ({'body': b'\n.', 'url': 'http://www.example.com'}, HtmlResponse), ] for source, cls in mappings: - retcls = responsetypes.from_args(**source) - assert retcls is cls, f"{source} ==> {retcls} != {cls}" + old_cls = _PRE_XTRACTMIME_RESPONSE_TYPES.from_args(**source) + new_cls = responsetypes.from_args(**source) + message = f"{source} ==> {old_cls} (old) != {new_cls} (current)" + assert old_cls == new_cls, message + assert new_cls is cls, f"{source} ==> {new_cls} != {cls}" def test_from_args_post_xtractmime(self): """Each of the following test cases got affected after @@ -138,12 +159,18 @@ class ResponseTypesTest(unittest.TestCase): ({'body': b'this is not {retcls} != {cls}" + old_cls = _PRE_XTRACTMIME_RESPONSE_TYPES.from_args(**source) + new_cls = responsetypes.from_args(**source) + message = f"{source} ==> {old_cls} (old) == {new_cls} (current)" + assert old_cls != new_cls, message + assert new_cls is cls, f"{source} ==> {new_cls} != {cls}" def test_custom_mime_types_loaded(self): - # check that mime.types files shipped with scrapy are loaded - self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test') + """Check that mime.types files shipped with Scrapy are loaded.""" + self.assertEqual( + _MIMETYPES.guess_type('x.scrapytest')[0], + 'x-scrapy/test', + ) if __name__ == "__main__": From 3fc9b732cb24764dbe9266a6a2e3540b3d839378 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 21 Jun 2022 11:12:10 +0200 Subject: [PATCH 29/81] Refactor responsetypes --- scrapy/responsetypes.py | 21 +++++++++++++------- tests/test_downloadermiddleware_httpcache.py | 2 +- tests/test_responsetypes.py | 12 ++++++++--- 3 files changed, 24 insertions(+), 11 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 77b190a98..13f9df52a 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -8,7 +8,11 @@ from io import StringIO from urllib.parse import urlparse from warnings import warn -from xtractmime import RESOURCE_HEADER_BUFFER_LENGTH, extract_mime, is_binary_data +from xtractmime import ( + RESOURCE_HEADER_BUFFER_LENGTH, + extract_mime, + is_binary_data, +) from xtractmime.mimegroups import ( is_html_mime_type, is_javascript_mime_type, @@ -22,13 +26,13 @@ from scrapy.utils.misc import load_object from scrapy.utils.python import binary_is_text, to_bytes, to_unicode -_CONTENT_ENCODING_MAP = { +_CONTENT_ENCODING_MIME_TYPES = { 'br': b'application/brotli', 'deflate': b'application/zip', 'gzip': b'application/gzip', } -_MIMETYPES = MimeTypes() -_MIMETYPES.readfp(StringIO(get_data('scrapy', 'mime.types').decode())) +_MIME_TYPES = MimeTypes() +_MIME_TYPES.readfp(StringIO(get_data('scrapy', 'mime.types').decode())) def _is_other_text_mime_type(mime_type): @@ -94,15 +98,17 @@ def _content_type_from_metadata(*, headers=None, url_path=None, filename=None): or url_mime_type ) + def _mime_type_from_path(path): - mimetype, encoding = _MIMETYPES.guess_type(path, strict=False) - encoding_mime_type = _CONTENT_ENCODING_MAP.get(encoding, None) + mimetype, encoding = _MIME_TYPES.guess_type(path, strict=False) + encoding_mime_type = _CONTENT_ENCODING_MIME_TYPES.get(encoding, None) if encoding_mime_type: return encoding_mime_type if mimetype: return mimetype.encode() return None + def _remove_nul_byte_from_text(text): """Return the text with removed null byte (b'\x00') if there are no other binary bytes in the text, otherwise return the text as-is. @@ -118,6 +124,7 @@ def _remove_nul_byte_from_text(text): return text.replace(b'\x00', b'') + def _response_type_from_mime_type(mime_type): if not mime_type: return Response @@ -204,7 +211,7 @@ class ResponseTypes: """Return the most appropriate Response class from a file name""" warn('ResponseTypes.from_filename is deprecated, ' 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) - mimetype, encoding = _MIMETYPES.guess_type(filename) + mimetype, encoding = _MIME_TYPES.guess_type(filename) if mimetype and not encoding: return self.from_mimetype(mimetype) else: diff --git a/tests/test_downloadermiddleware_httpcache.py b/tests/test_downloadermiddleware_httpcache.py index 7945d2a61..928c007f5 100644 --- a/tests/test_downloadermiddleware_httpcache.py +++ b/tests/test_downloadermiddleware_httpcache.py @@ -128,7 +128,7 @@ class DefaultStorageTest(_BaseTest): with self._storage() as storage: assert storage.retrieve_response(self.spider, self.request) is None response = Response( - 'http://www.example.com/', + 'http://www.example.com', body=b'\n.', status=202, ) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 3263dcb24..ebfe78f1e 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -1,7 +1,13 @@ import unittest -from scrapy.responsetypes import _MIMETYPES, responsetypes, ResponseTypes -from scrapy.http import Response, TextResponse, XmlResponse, HtmlResponse, Headers +from scrapy.http import ( + Headers, + HtmlResponse, + Response, + TextResponse, + XmlResponse, +) +from scrapy.responsetypes import _MIME_TYPES, responsetypes, ResponseTypes class PreXtractmimeResponseTypes(ResponseTypes): @@ -168,7 +174,7 @@ class ResponseTypesTest(unittest.TestCase): def test_custom_mime_types_loaded(self): """Check that mime.types files shipped with Scrapy are loaded.""" self.assertEqual( - _MIMETYPES.guess_type('x.scrapytest')[0], + _MIME_TYPES.guess_type('x.scrapytest')[0], 'x-scrapy/test', ) From d2accf2e39dba6d6d48eeb9683081067fa46a352 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 21 Jun 2022 12:02:56 +0200 Subject: [PATCH 30/81] Fix issues reported by static checks --- scrapy/responsetypes.py | 3 ++- tests/test_responsetypes.py | 4 ++-- 2 files changed, 4 insertions(+), 3 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 13f9df52a..1bf21ca81 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -32,7 +32,8 @@ _CONTENT_ENCODING_MIME_TYPES = { 'gzip': b'application/gzip', } _MIME_TYPES = MimeTypes() -_MIME_TYPES.readfp(StringIO(get_data('scrapy', 'mime.types').decode())) +_scrapy_mime_data = get_data('scrapy', 'mime.types') or b'' +_MIME_TYPES.readfp(StringIO(_scrapy_mime_data.decode())) def _is_other_text_mime_type(mime_type): diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index ebfe78f1e..075b0ad22 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -174,8 +174,8 @@ class ResponseTypesTest(unittest.TestCase): def test_custom_mime_types_loaded(self): """Check that mime.types files shipped with Scrapy are loaded.""" self.assertEqual( - _MIME_TYPES.guess_type('x.scrapytest')[0], - 'x-scrapy/test', + _MIME_TYPES.guess_type('x.scrapytest')[0], + 'x-scrapy/test', ) From 19a00bd4ef58ed38d0c3c8e7b24a9f2cd45f65f5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 21 Jun 2022 19:07:17 +0200 Subject: [PATCH 31/81] Make xtractmime a dependency --- setup.py | 1 + tox.ini | 3 +-- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/setup.py b/setup.py index ed197273f..024df712e 100644 --- a/setup.py +++ b/setup.py @@ -34,6 +34,7 @@ install_requires = [ 'setuptools', 'tldextract', 'lxml>=4.3.0', + 'xtractmime>=0.1.0', ] extras_require = {} cpython_dependencies = [ diff --git a/tox.ini b/tox.ini index dc561f270..a65395a6e 100644 --- a/tox.ini +++ b/tox.ini @@ -18,7 +18,6 @@ deps = mitmproxy >= 4.0.4, < 8; python_version < '3.9' and implementation_name != 'pypy' # newer markupsafe is incompatible with deps of old mitmproxy (which we get on Python 3.7 and lower) markupsafe < 2.1.0; python_version < '3.8' and implementation_name != 'pypy' - git+https://github.com/scrapy/xtractmime.git@main#egg=xtractmime # Extras botocore>=1.4.87 passenv = @@ -86,7 +85,7 @@ deps = w3lib==1.17.0 zope.interface==5.1.0 lxml==4.3.0 - git+https://github.com/scrapy/xtractmime.git@main#egg=xtractmime + xtractmime==0.1.0 -rtests/requirements.txt # mitmproxy 4.0.4+ requires upgrading some of the pinned dependencies From fd2317bd6de8810f6cf9b9900c19fea40b171679 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 15:15:52 +0200 Subject: [PATCH 32/81] Implement scrapy.utils.response.get_response_class --- scrapy/core/downloader/handlers/datauri.py | 7 +- scrapy/core/downloader/handlers/file.py | 4 +- scrapy/core/downloader/handlers/ftp.py | 4 +- scrapy/core/downloader/handlers/http11.py | 4 +- scrapy/core/downloader/webclient.py | 4 +- scrapy/core/http2/stream.py | 6 +- scrapy/downloadermiddlewares/decompression.py | 10 +- .../downloadermiddlewares/httpcompression.py | 8 +- scrapy/extensions/httpcache.py | 6 +- scrapy/http/request/form.py | 3 +- scrapy/http/response/text.py | 19 +- scrapy/linkextractors/lxmlhtml.py | 17 +- scrapy/responsetypes.py | 174 ++----------- scrapy/utils/response.py | 186 +++++++++++++- tests/test_downloader_handlers.py | 7 +- ...st_downloadermiddleware_httpcompression.py | 8 +- tests/test_http_response.py | 14 +- tests/test_responsetypes.py | 132 +++------- tests/test_utils_response.py | 238 ++++++++++++++++-- tox.ini | 5 +- 20 files changed, 540 insertions(+), 316 deletions(-) diff --git a/scrapy/core/downloader/handlers/datauri.py b/scrapy/core/downloader/handlers/datauri.py index a45b4ff3c..3c09dd246 100644 --- a/scrapy/core/downloader/handlers/datauri.py +++ b/scrapy/core/downloader/handlers/datauri.py @@ -1,8 +1,8 @@ from w3lib.url import parse_data_uri from scrapy.http import TextResponse -from scrapy.responsetypes import responsetypes from scrapy.utils.decorators import defers +from scrapy.utils.response import get_response_class class DataURIDownloadHandler: @@ -11,7 +11,10 @@ class DataURIDownloadHandler: @defers def download_request(self, request, spider): uri = parse_data_uri(request.url) - respcls = responsetypes.from_mimetype(uri.media_type) + respcls = get_response_class( + body=uri.data, + declared_mime_types=(uri.media_type.encode(),), + ) resp_kwargs = {} if (issubclass(respcls, TextResponse) diff --git a/scrapy/core/downloader/handlers/file.py b/scrapy/core/downloader/handlers/file.py index 0d94e3df0..ffff915fa 100644 --- a/scrapy/core/downloader/handlers/file.py +++ b/scrapy/core/downloader/handlers/file.py @@ -1,7 +1,7 @@ from w3lib.url import file_uri_to_path -from scrapy.responsetypes import responsetypes from scrapy.utils.decorators import defers +from scrapy.utils.response import get_response_class class FileDownloadHandler: @@ -12,5 +12,5 @@ class FileDownloadHandler: filepath = file_uri_to_path(request.url) with open(filepath, 'rb') as fo: body = fo.read() - respcls = responsetypes.from_args(filename=filepath, body=body) + respcls = get_response_class(url=request.url, body=body) return respcls(url=request.url, body=body) diff --git a/scrapy/core/downloader/handlers/ftp.py b/scrapy/core/downloader/handlers/ftp.py index a495874bd..397ff7b98 100644 --- a/scrapy/core/downloader/handlers/ftp.py +++ b/scrapy/core/downloader/handlers/ftp.py @@ -36,9 +36,9 @@ from twisted.internet.protocol import ClientCreator, Protocol from twisted.protocols.ftp import CommandFailed, FTPClient from scrapy.http import Response -from scrapy.responsetypes import responsetypes from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.python import to_bytes +from scrapy.utils.response import get_response_class class ReceivedDataProtocol(Protocol): @@ -105,7 +105,7 @@ class FTPDownloadHandler: protocol.close() headers = {"local filename": protocol.filename or '', "size": protocol.size} body = to_bytes(protocol.filename or protocol.body.read()) - respcls = responsetypes.from_args(url=request.url, body=body) + respcls = get_response_class(url=request.url, body=body) return respcls(url=request.url, status=200, body=body, headers=headers) def _failed(self, result, request): diff --git a/scrapy/core/downloader/handlers/http11.py b/scrapy/core/downloader/handlers/http11.py index 38935667d..c40389aae 100644 --- a/scrapy/core/downloader/handlers/http11.py +++ b/scrapy/core/downloader/handlers/http11.py @@ -24,8 +24,8 @@ from scrapy.core.downloader.contextfactory import load_context_factory_from_sett from scrapy.core.downloader.webclient import _parse from scrapy.exceptions import ScrapyDeprecationWarning, StopDownload from scrapy.http import Headers -from scrapy.responsetypes import responsetypes from scrapy.utils.python import to_bytes, to_unicode +from scrapy.utils.response import get_response_class logger = logging.getLogger(__name__) @@ -449,7 +449,7 @@ class ScrapyAgent: def _cb_bodydone(self, result, request, url): headers = self._headers_from_twisted_response(result["txresponse"]) - respcls = responsetypes.from_args(headers=headers, url=url, body=result["body"]) + respcls = get_response_class(http_headers=headers, url=url, body=result["body"]) try: version = result["txresponse"].version protocol = f"{to_unicode(version[0])}/{version[1]}.{version[2]}" diff --git a/scrapy/core/downloader/webclient.py b/scrapy/core/downloader/webclient.py index 7d048c1e4..a97ef7027 100644 --- a/scrapy/core/downloader/webclient.py +++ b/scrapy/core/downloader/webclient.py @@ -9,7 +9,7 @@ from twisted.internet.protocol import ClientFactory from scrapy.http import Headers from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.python import to_bytes, to_unicode -from scrapy.responsetypes import responsetypes +from scrapy.utils.response import get_response_class def _parsed_url_args(parsed): @@ -112,7 +112,7 @@ class ScrapyHTTPClientFactory(ClientFactory): request.meta['download_latency'] = self.headers_time - self.start_time status = int(self.status) headers = Headers(self.response_headers) - respcls = responsetypes.from_args(headers=headers, url=self._url, body=body) + respcls = get_response_class(http_headers=headers, url=self._url, body=body) return respcls(url=self._url, status=status, headers=headers, body=body, protocol=to_unicode(self.version)) def _set_connection_attributes(self, request): diff --git a/scrapy/core/http2/stream.py b/scrapy/core/http2/stream.py index 5c393c027..a36c8c36c 100644 --- a/scrapy/core/http2/stream.py +++ b/scrapy/core/http2/stream.py @@ -14,7 +14,7 @@ from twisted.web.client import ResponseFailed from scrapy.http import Request from scrapy.http.headers import Headers -from scrapy.responsetypes import responsetypes +from scrapy.utils.response import get_response_class if TYPE_CHECKING: from scrapy.core.http2.protocol import H2ClientProtocol @@ -450,8 +450,8 @@ class Stream: generated response instance""" body = self._response['body'].getvalue() - response_cls = responsetypes.from_args( - headers=self._response['headers'], + response_cls = get_response_class( + http_headers=self._response['headers'], url=self._request.url, body=body, ) diff --git a/scrapy/downloadermiddlewares/decompression.py b/scrapy/downloadermiddlewares/decompression.py index 0fcf8fb8c..389755d12 100644 --- a/scrapy/downloadermiddlewares/decompression.py +++ b/scrapy/downloadermiddlewares/decompression.py @@ -10,7 +10,7 @@ import zipfile from io import BytesIO from tempfile import mktemp -from scrapy.responsetypes import responsetypes +from scrapy.utils.response import get_response_class logger = logging.getLogger(__name__) @@ -36,7 +36,7 @@ class DecompressionMiddleware: return body = tar_file.extractfile(tar_file.members[0]).read() - respcls = responsetypes.from_args(filename=tar_file.members[0].name, body=body) + respcls = get_response_class(url=tar_file.members[0].name, body=body) return response.replace(body=body, cls=respcls) def _is_zip(self, response): @@ -48,7 +48,7 @@ class DecompressionMiddleware: namelist = zip_file.namelist() body = zip_file.read(namelist[0]) - respcls = responsetypes.from_args(filename=namelist[0], body=body) + respcls = get_response_class(url=namelist[0], body=body) return response.replace(body=body, cls=respcls) def _is_gzip(self, response): @@ -58,7 +58,7 @@ class DecompressionMiddleware: except IOError: return - respcls = responsetypes.from_args(body=body) + respcls = get_response_class(body=body) return response.replace(body=body, cls=respcls) def _is_bzip2(self, response): @@ -67,7 +67,7 @@ class DecompressionMiddleware: except IOError: return - respcls = responsetypes.from_args(body=body) + respcls = get_response_class(body=body) return response.replace(body=body, cls=respcls) def process_response(self, request, response, spider): diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 4e7feeeaf..11407ca46 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -4,9 +4,9 @@ import zlib from scrapy.exceptions import NotConfigured from scrapy.http import Response, TextResponse -from scrapy.responsetypes import responsetypes from scrapy.utils.deprecate import ScrapyDeprecationWarning from scrapy.utils.gz import gunzip +from scrapy.utils.response import get_response_class ACCEPTED_ENCODINGS = [b'gzip', b'deflate'] @@ -63,8 +63,10 @@ class HttpCompressionMiddleware: if self.stats: self.stats.inc_value('httpcompression/response_bytes', len(decoded_body), spider=spider) self.stats.inc_value('httpcompression/response_count', spider=spider) - respcls = responsetypes.from_args( - headers=response.headers, url=response.url, body=decoded_body + respcls = get_response_class( + http_headers=response.headers, + url=response.url, + body=decoded_body, ) kwargs = dict(cls=respcls, body=decoded_body) if issubclass(respcls, TextResponse): diff --git a/scrapy/extensions/httpcache.py b/scrapy/extensions/httpcache.py index 843e14812..a5fe87232 100644 --- a/scrapy/extensions/httpcache.py +++ b/scrapy/extensions/httpcache.py @@ -10,10 +10,10 @@ from weakref import WeakKeyDictionary from w3lib.http import headers_raw_to_dict, headers_dict_to_raw from scrapy.http import Headers, Response -from scrapy.responsetypes import responsetypes from scrapy.utils.httpobj import urlparse_cached from scrapy.utils.project import data_path from scrapy.utils.python import to_bytes, to_unicode +from scrapy.utils.response import get_response_class logger = logging.getLogger(__name__) @@ -240,7 +240,7 @@ class DbmCacheStorage: status = data['status'] headers = Headers(data['headers']) body = data['body'] - respcls = responsetypes.from_args(headers=headers, url=url, body=body) + respcls = get_response_class(http_headers=headers, url=url, body=body) response = respcls(url=url, headers=headers, status=status, body=body) return response @@ -299,7 +299,7 @@ class FilesystemCacheStorage: url = metadata.get('response_url') status = metadata['status'] headers = Headers(headers_raw_to_dict(rawheaders)) - respcls = responsetypes.from_args(headers=headers, url=url, body=body) + respcls = get_response_class(http_headers=headers, url=url, body=body) response = respcls(url=url, headers=headers, status=status, body=body) return response diff --git a/scrapy/http/request/form.py b/scrapy/http/request/form.py index 0c947565a..77341962f 100644 --- a/scrapy/http/request/form.py +++ b/scrapy/http/request/form.py @@ -15,7 +15,6 @@ from w3lib.html import strip_html5_whitespace from scrapy.http.request import Request from scrapy.http.response.text import TextResponse from scrapy.utils.python import to_bytes, is_listlike -from scrapy.utils.response import get_base_url FormRequestTypeVar = TypeVar("FormRequestTypeVar", bound="FormRequest") @@ -98,7 +97,7 @@ def _get_form( formxpath: Optional[str], ) -> FormElement: """Find the wanted form element within the given response.""" - root = create_root_node(response.text, HTMLParser, base_url=get_base_url(response)) + root = create_root_node(response.text, HTMLParser, base_url=response.base_url) forms = root.xpath('//form') if not forms: raise ValueError(f"No
element found in {response}") diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 89516b9b6..e1cae274f 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -13,12 +13,11 @@ from urllib.parse import urljoin import parsel from w3lib.encoding import (html_body_declared_encoding, html_to_unicode, http_content_type_encoding, resolve_encoding) -from w3lib.html import strip_html5_whitespace +from w3lib.html import get_base_url, strip_html5_whitespace from scrapy.http import Request from scrapy.http.response import Response from scrapy.utils.python import memoizemethod_noargs, to_unicode -from scrapy.utils.response import get_base_url _NONE = object() @@ -32,6 +31,7 @@ class TextResponse(Response): def __init__(self, *args, **kwargs): self._encoding = kwargs.pop('encoding', None) + self._cached_base_url = None self._cached_benc = None self._cached_ubody = None self._cached_selector = None @@ -42,6 +42,7 @@ class TextResponse(Response): self._url = to_unicode(url, self.encoding) else: super()._set_url(url) + self._cached_base_url = None def _set_body(self, body): self._body = b'' # used by encoding detection @@ -52,6 +53,7 @@ class TextResponse(Response): self._body = body.encode(self._encoding) else: super()._set_body(body) + self._cached_base_url = None @property def encoding(self): @@ -85,10 +87,21 @@ class TextResponse(Response): self._cached_ubody = html_to_unicode(charset, self.body)[1] return self._cached_ubody + @property + def base_url(self) -> str: + """Base URL""" + if self._cached_base_url is None: + self._cached_base_url = get_base_url( + self.text[:4096], + self.url, + self.encoding, + ) + return self._cached_base_url + def urljoin(self, url): """Join this Response's url with a possible relative url to form an absolute interpretation of the latter.""" - return urljoin(get_base_url(self), url) + return urljoin(self.base_url, url) @memoizemethod_noargs def _headers_encoding(self): diff --git a/scrapy/linkextractors/lxmlhtml.py b/scrapy/linkextractors/lxmlhtml.py index b5d2585a8..6c41758a1 100644 --- a/scrapy/linkextractors/lxmlhtml.py +++ b/scrapy/linkextractors/lxmlhtml.py @@ -13,7 +13,6 @@ from scrapy.link import Link from scrapy.linkextractors import FilteringLinkExtractor from scrapy.utils.misc import arg_to_iter, rel_has_nofollow from scrapy.utils.python import unique as unique_list -from scrapy.utils.response import get_base_url # from lxml/src/lxml/html/__init__.py @@ -82,8 +81,12 @@ class LxmlParserLinkExtractor: return self._deduplicate_if_needed(links) def extract_links(self, response): - base_url = get_base_url(response) - return self._extract_links(response.selector, response.url, response.encoding, base_url) + return self._extract_links( + response.selector, + response.url, + response.encoding, + response.base_url, + ) def _process_links(self, links): """ Normalize and filter extracted links @@ -148,7 +151,6 @@ class LxmlLinkExtractor(FilteringLinkExtractor): Duplicate links are omitted. """ - base_url = get_base_url(response) if self.restrict_xpaths: docs = [ subdoc @@ -159,6 +161,11 @@ class LxmlLinkExtractor(FilteringLinkExtractor): docs = [response.selector] all_links = [] for doc in docs: - links = self._extract_links(doc, response.url, response.encoding, base_url) + links = self._extract_links( + doc, + response.url, + response.encoding, + response.base_url, + ) all_links.extend(self._process_links(links)) return unique_list(all_links) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 1bf21ca81..7308e0350 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -2,140 +2,13 @@ This module implements a class which returns the appropriate Response class based on different criteria. """ -from mimetypes import MimeTypes -from pkgutil import get_data -from io import StringIO -from urllib.parse import urlparse from warnings import warn -from xtractmime import ( - RESOURCE_HEADER_BUFFER_LENGTH, - extract_mime, - is_binary_data, -) -from xtractmime.mimegroups import ( - is_html_mime_type, - is_javascript_mime_type, - is_json_mime_type, - is_xml_mime_type, -) - from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse +from scrapy.http import Response from scrapy.utils.misc import load_object from scrapy.utils.python import binary_is_text, to_bytes, to_unicode - - -_CONTENT_ENCODING_MIME_TYPES = { - 'br': b'application/brotli', - 'deflate': b'application/zip', - 'gzip': b'application/gzip', -} -_MIME_TYPES = MimeTypes() -_scrapy_mime_data = get_data('scrapy', 'mime.types') or b'' -_MIME_TYPES.readfp(StringIO(_scrapy_mime_data.decode())) - - -def _is_other_text_mime_type(mime_type): - value = ( - mime_type.startswith(b'text/') - or is_json_mime_type(mime_type) - or is_javascript_mime_type(mime_type) - or mime_type in ( - b'application/x-json', - b'application/json-amazonui-streaming', - b'application/x-javascript', - ) - ) - return value - - -_PRIORITIZED_MIME_TYPE_CHECKERS = ( - is_html_mime_type, - is_xml_mime_type, - _is_other_text_mime_type, -) - - -def _content_type_from_metadata(*, headers=None, url_path=None, filename=None): - if headers and b'Content-Type' in headers: - content_type_mime_type = headers.getlist(b'Content-Type')[-1] - else: - content_type_mime_type = None - - if headers and b'Content-Disposition' in headers: - _filename = ( - headers.get(b"Content-Disposition") - .split(b";")[-1] - .split(b"=")[-1] - .strip(b"\"'") - .decode() - ) - content_disposition_mime_type = _mime_type_from_path(_filename) - else: - content_disposition_mime_type = None - - filename_mime_type = _mime_type_from_path(filename) if filename else None - url_mime_type = _mime_type_from_path(url_path) if url_path else None - - candidate_mime_types = tuple( - mime_type - for mime_type in ( - content_type_mime_type, - content_disposition_mime_type, - filename_mime_type, - url_mime_type, - ) - if mime_type is not None - ) - for mime_type_checker in _PRIORITIZED_MIME_TYPE_CHECKERS: - for candidate_mime_type in candidate_mime_types: - if mime_type_checker(candidate_mime_type): - return candidate_mime_type - return ( - content_type_mime_type - or content_disposition_mime_type - or filename_mime_type - or url_mime_type - ) - - -def _mime_type_from_path(path): - mimetype, encoding = _MIME_TYPES.guess_type(path, strict=False) - encoding_mime_type = _CONTENT_ENCODING_MIME_TYPES.get(encoding, None) - if encoding_mime_type: - return encoding_mime_type - if mimetype: - return mimetype.encode() - return None - - -def _remove_nul_byte_from_text(text): - """Return the text with removed null byte (b'\x00') if there are no other - binary bytes in the text, otherwise return the text as-is. - - Based on https://github.com/scrapy/scrapy/issues/2481 - """ - for index in range(len(text)): - if ( - text[index:index + 1] != b'\x00' - and is_binary_data(text[index:index + 1]) - ): - return text - - return text.replace(b'\x00', b'') - - -def _response_type_from_mime_type(mime_type): - if not mime_type: - return Response - if is_html_mime_type(mime_type): - return HtmlResponse - if is_xml_mime_type(mime_type): - return XmlResponse - if _is_other_text_mime_type(mime_type): - return TextResponse - return Response +from scrapy.utils.response import _MIME_TYPES class ResponseTypes: @@ -159,11 +32,14 @@ class ResponseTypes: def __init__(self): self.classes = {} + self.mimetypes = _MIME_TYPES for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" + warn('ResponseTypes.from_mimetype is deprecated, ' + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) if mimetype is None: return Response elif mimetype in self.classes: @@ -176,7 +52,7 @@ class ResponseTypes: """Return the most appropriate Response class from an HTTP Content-Type header """ warn('ResponseTypes.from_content_type is deprecated, ' - 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) if content_encoding: return Response mimetype = to_unicode(content_type).split(';')[0].strip().lower() @@ -184,7 +60,7 @@ class ResponseTypes: def from_content_disposition(self, content_disposition): warn('ResponseTypes.from_content_disposition is deprecated, ' - 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) try: filename = to_unicode( content_disposition, encoding='latin-1', errors='replace' @@ -197,7 +73,7 @@ class ResponseTypes: """Return the most appropriate Response class by looking at the HTTP headers""" warn('ResponseTypes.from_headers is deprecated, ' - 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) cls = Response if b'Content-Type' in headers: cls = self.from_content_type( @@ -211,8 +87,8 @@ class ResponseTypes: def from_filename(self, filename): """Return the most appropriate Response class from a file name""" warn('ResponseTypes.from_filename is deprecated, ' - 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) - mimetype, encoding = _MIME_TYPES.guess_type(filename) + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) + mimetype, encoding = self.mimetypes.guess_type(filename) if mimetype and not encoding: return self.from_mimetype(mimetype) else: @@ -224,7 +100,7 @@ class ResponseTypes: it's not meant to be used except for special cases where response types cannot be guess using more straightforward methods.""" warn('ResponseTypes.from_body is deprecated, ' - 'please use ResponseTypes.from_args instead', ScrapyDeprecationWarning) + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) chunk = body[:5000] chunk = to_bytes(chunk) if not binary_is_text(chunk): @@ -241,22 +117,18 @@ class ResponseTypes: def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on the given arguments.""" - body = body or b'' - body = _remove_nul_byte_from_text(body[:RESOURCE_HEADER_BUFFER_LENGTH]) - url_parts = urlparse(url) if url else url - content_type = _content_type_from_metadata( - headers=headers, - url_path=url_parts.path if url_parts else url_parts, - filename=filename, - ) - content_types = (content_type,) if content_type else None - http_origin = not url or url_parts.scheme in ("http", "https") - mime_type = extract_mime( - body, - content_types=content_types, - http_origin=http_origin, - ) - return _response_type_from_mime_type(mime_type) + warn('ResponseTypes.from_args is deprecated, ' + 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) + cls = Response + if headers is not None: + cls = self.from_headers(headers) + if cls is Response and url is not None: + cls = self.from_filename(url) + if cls is Response and filename is not None: + cls = self.from_filename(filename) + if cls is Response and body is not None: + cls = self.from_body(body) + return cls responsetypes = ResponseTypes() diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 741dce350..6d1b575b8 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -6,27 +6,159 @@ import os import re import tempfile import webbrowser -from typing import Any, Callable, Iterable, Optional, Tuple, Union +from io import StringIO +from mimetypes import MimeTypes +from pkgutil import get_data +from typing import ( + Any, + Callable, + Iterable, + Optional, + Sequence, + Tuple, + Type, + Union, +) +from urllib.parse import urlparse +from warnings import warn from weakref import WeakKeyDictionary -import scrapy -from scrapy.http.response import Response - from twisted.web import http -from scrapy.utils.python import to_bytes, to_unicode -from scrapy.utils.decorators import deprecated from w3lib import html +from xtractmime import ( + RESOURCE_HEADER_BUFFER_LENGTH as BODY_LIMIT, + extract_mime, + is_binary_data, +) +from xtractmime.mimegroups import ( + is_html_mime_type, + is_javascript_mime_type, + is_json_mime_type, + is_xml_mime_type, +) + +import scrapy +from scrapy.exceptions import ScrapyDeprecationWarning +from scrapy.http import ( + Headers, + HtmlResponse, + Response, + TextResponse, + XmlResponse, +) +from scrapy.utils.decorators import deprecated +from scrapy.utils.python import to_bytes, to_unicode _baseurl_cache: "WeakKeyDictionary[Response, str]" = WeakKeyDictionary() +_CONTENT_ENCODING_MIME_TYPES = { + 'br': b'application/brotli', + 'deflate': b'application/zip', + 'gzip': b'application/gzip', +} +_MIME_TYPES = MimeTypes() +_mime_overrides = get_data('scrapy', 'mime.types') or b'' +_MIME_TYPES.readfp(StringIO(_mime_overrides.decode())) -def get_base_url(response: "scrapy.http.response.text.TextResponse") -> str: +def _is_other_text_mime_type(mime_type): + return ( + mime_type.startswith(b'text/') + or is_json_mime_type(mime_type) + or is_javascript_mime_type(mime_type) + or mime_type in ( + b'application/x-json', + b'application/json-amazonui-streaming', + b'application/x-javascript', + ) + ) + + +_PRIORITIZED_MIME_TYPE_CHECKERS = ( + is_html_mime_type, + is_xml_mime_type, + _is_other_text_mime_type, +) + + +def _get_best_mime_type(mime_types): + candidate_mime_types = tuple( + mime_type + for mime_type in mime_types + if mime_type is not None + ) + for mime_type_checker in _PRIORITIZED_MIME_TYPE_CHECKERS: + for candidate_mime_type in candidate_mime_types: + if mime_type_checker(candidate_mime_type): + return candidate_mime_type + return mime_types[0] + + +def _get_http_header_mime_types(headers: Headers) -> Sequence[bytes]: + mime_types = [] + if b'Content-Type' in headers: + mime_types.append(headers[b'Content-Type'].split(b';')[0]) + if b'Content-Disposition' in headers: + path = ( + headers.get(b"Content-Disposition") + .split(b";")[-1] + .split(b"=")[-1] + .strip(b"\"'") + .decode() + ) + mime_types.append(_get_mime_type_from_path(path)) + return mime_types + + +def _get_mime_type_from_path(path): + mimetype, encoding = _MIME_TYPES.guess_type(path, strict=False) + encoding_mime_type = _CONTENT_ENCODING_MIME_TYPES.get(encoding, None) + if encoding_mime_type: + return encoding_mime_type + if mimetype: + return mimetype.encode() + return None + + +def _get_response_class_from_mime_type(mime_type): + if not mime_type: + return Response + if is_html_mime_type(mime_type): + return HtmlResponse + if is_xml_mime_type(mime_type): + return XmlResponse + if _is_other_text_mime_type(mime_type): + return TextResponse + return Response + + +def _remove_nul_byte_from_text(text): + """Return the text with removed null byte (b'\x00') if there are no other + binary bytes in the text, otherwise return the text as-is. + + Based on https://github.com/scrapy/scrapy/issues/2481 + """ + for index in range(len(text)): + if ( + text[index:index + 1] != b'\x00' + and is_binary_data(text[index:index + 1]) + ): + return text + + return text.replace(b'\x00', b'') + + +def get_base_url(response: TextResponse) -> str: """Return the base url of the given response, joined with the response url""" - if response not in _baseurl_cache: - text = response.text[0:4096] - _baseurl_cache[response] = html.get_base_url(text, response.url, response.encoding) - return _baseurl_cache[response] + warn( + ( + "scrapy.utils.response.get_base_url is deprecated, use " + "scrapy.http.TextResponse.base_url instead." + ), + ScrapyDeprecationWarning, + stacklevel=2, + ) + return response.base_url _metaref_cache: "WeakKeyDictionary[Response, Union[Tuple[None, None], Tuple[float, str]]]" = WeakKeyDictionary() @@ -44,6 +176,38 @@ def get_meta_refresh( return _metaref_cache[response] +def get_response_class( + *, + url: str = None, + body: bytes = None, + declared_mime_types: Sequence[bytes] = None, + http_headers: Headers = None, +) -> Type[Response]: + """Guess the most appropriate Response class based on the given + arguments.""" + mime_types = list(declared_mime_types or []) + if http_headers: + mime_types.extend(_get_http_header_mime_types(http_headers)) + if url is not None: + url_parts = urlparse(url) + http_origin = url_parts.scheme in ("http", "https") + mime_types.append(_get_mime_type_from_path(url_parts.path)) + else: + http_origin = True + body = _remove_nul_byte_from_text((body or b'')[:BODY_LIMIT]) + if mime_types: + best_mime_type = _get_best_mime_type(mime_types) + content_types = (best_mime_type,) if best_mime_type else best_mime_type + else: + content_types = None + mime_type = extract_mime( + body, + content_types=content_types, + http_origin=http_origin, + ) + return _get_response_class_from_mime_type(mime_type) + + def response_status_message(status: Union[bytes, float, int, str]) -> str: """Return status code plus status text descriptive message """ diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 72f52121e..e9f890391 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -27,7 +27,6 @@ from scrapy.core.downloader.handlers.s3 import S3DownloadHandler from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning from scrapy.http import Headers, HtmlResponse, Request from scrapy.http.response.text import TextResponse -from scrapy.responsetypes import responsetypes from scrapy.spiders import Spider from scrapy.utils.misc import create_instance from scrapy.utils.python import to_bytes @@ -1190,7 +1189,7 @@ class DataURITestCase(unittest.TestCase): def test_default_mediatype_encoding(self): def _test(response): self.assertEqual(response.text, 'A brief note') - self.assertEqual(type(response), responsetypes.from_mimetype("text/plain")) + self.assertIsInstance(response, TextResponse) self.assertEqual(response.encoding, "US-ASCII") request = Request("data:,A%20brief%20note") @@ -1199,7 +1198,7 @@ class DataURITestCase(unittest.TestCase): def test_default_mediatype(self): def _test(response): self.assertEqual(response.text, '\u038e\u03a3\u038e') - self.assertEqual(type(response), responsetypes.from_mimetype("text/plain")) + self.assertIsInstance(response, TextResponse) self.assertEqual(response.encoding, "iso-8859-7") request = Request("data:;charset=iso-8859-7,%be%d3%be") @@ -1217,7 +1216,7 @@ class DataURITestCase(unittest.TestCase): def test_mediatype_parameters(self): def _test(response): self.assertEqual(response.text, '\u038e\u03a3\u038e') - self.assertEqual(type(response), responsetypes.from_mimetype("text/plain")) + self.assertIsInstance(response, TextResponse) self.assertEqual(response.encoding, "utf-8") request = Request('data:text/plain;foo=%22foo;bar%5C%22%22;' diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index 40e9f3a96..4a670742a 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -8,8 +8,8 @@ from scrapy.spiders import Spider from scrapy.http import Response, Request, HtmlResponse from scrapy.downloadermiddlewares.httpcompression import HttpCompressionMiddleware, ACCEPTED_ENCODINGS from scrapy.exceptions import NotConfigured, ScrapyDeprecationWarning -from scrapy.responsetypes import responsetypes from scrapy.utils.gz import gunzip +from scrapy.utils.response import get_response_class from scrapy.utils.test import get_crawler from tests import tests_datadir from w3lib.encoding import resolve_encoding @@ -247,7 +247,11 @@ class HttpCompressionTest(TestCase): } plainbody = (b'Some page' b'') - respcls = responsetypes.from_args(url="http://www.example.com/index", headers=headers, body=plainbody) + respcls = get_response_class( + url="http://www.example.com/index", + http_headers=headers, + body=plainbody, + ) response = respcls("http://www.example.com/index", headers=headers, body=plainbody) request = Request("http://www.example.com/index") diff --git a/tests/test_http_response.py b/tests/test_http_response.py index 2986f884f..05f2380e9 100644 --- a/tests/test_http_response.py +++ b/tests/test_http_response.py @@ -500,7 +500,7 @@ class TextResponseTest(BaseResponseTest): ) def test_urljoin_with_base_url(self): - """Test urljoin shortcut which also evaluates base-url through get_base_url().""" + """Test urljoin shortcut which also evaluates base-url.""" body = b'' joined = self.response_class('http://www.example.com', body=body).urljoin('/test') absolute = 'https://example.net/test' @@ -697,6 +697,18 @@ class HtmlResponseTest(TextResponseTest): response_class = HtmlResponse + def test_base_url(self): + resp = HtmlResponse("http://www.example.com", body=b""" + + + blahablsdfsal& + """) + self.assertEqual(resp.base_url, "http://www.example.com/img/") + + resp2 = HtmlResponse("http://www.example.com", body=b""" + blahablsdfsal&""") + self.assertEqual(resp2.base_url, "http://www.example.com") + def test_html_encoding(self): body = b"""Some page diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 075b0ad22..3d5217385 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -1,5 +1,7 @@ import unittest +import pytest + from scrapy.http import ( Headers, HtmlResponse, @@ -7,24 +9,37 @@ from scrapy.http import ( TextResponse, XmlResponse, ) -from scrapy.responsetypes import _MIME_TYPES, responsetypes, ResponseTypes +from scrapy.responsetypes import responsetypes +from .test_utils_response import ( + POST_XTRACTMIME_SCENARIOS, + PRE_XTRACTMIME_SCENARIOS, +) -class PreXtractmimeResponseTypes(ResponseTypes): - def from_args(self, headers=None, url=None, filename=None, body=None): - cls = Response - if headers is not None: - cls = self.from_headers(headers) - if cls is Response and url is not None: - cls = self.from_filename(url) - if cls is Response and filename is not None: - cls = self.from_filename(filename) - if cls is Response and body is not None: - cls = self.from_body(body) - return cls - - -_PRE_XTRACTMIME_RESPONSE_TYPES = PreXtractmimeResponseTypes() +@pytest.mark.parametrize( + "kwargs,response_class", + ( + *PRE_XTRACTMIME_SCENARIOS, + *( + pytest.param( + kwargs, + response_class, + marks=pytest.mark.xfail( + strict=True, + reason=( + "Expected failure of deprecated " + "scrapy.responsetypes.responsetypes.from_args, works " + "with its replacement " + "scrapy.utils.response.get_response_class" + ), + ), + ) + for kwargs, response_class in POST_XTRACTMIME_SCENARIOS + ), + ), +) +def test_from_args(kwargs, response_class): + assert responsetypes.from_args(**kwargs) == response_class class ResponseTypesTest(unittest.TestCase): @@ -77,6 +92,8 @@ class ResponseTypesTest(unittest.TestCase): (b'\x03\x02\xdf\xdd\x23', Response), (b'Some plain text\ndata with tabs\t and null bytes\0', TextResponse), (b'Hello', HtmlResponse), + # https://codersblock.com/blog/the-smallest-valid-html5-page/ + (b'\n.', HtmlResponse), (b' {retcls} != {cls}" - def test_from_args_pre_xtractmime(self): - """Each of the following test cases remains unaffected after - using xtractmime for MIME sniffing""" - mappings = [ - ({'url': 'http://www.example.com/data.csv'}, TextResponse), - ({'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), - 'url': 'http://www.example.com/item/'}, HtmlResponse), - ({'body': b'\x01\x02', 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), - 'url': 'http://www.example.com/page/'}, Response), - ({'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, TextResponse), - ({'body': b'\x03\x02\xdf\xdd\x23', 'headers': Headers({'Content-Encoding': 'UTF-8'})}, - Response), - ({'body': b'\x00\x01\xff', 'url': '://www.example.com/item/', - 'headers': Headers({'Content-Type': ['text/plain']})}, TextResponse), - ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), - ({'body': b'Hello'}, HtmlResponse), - ({'body': b'\n.'}, HtmlResponse), - ({'body': b'\x01\x02', 'headers': Headers({'Content-Type': ['application/pdf']})}, Response), - ({'headers': Headers({'Content-Type': ['application/x-json']})}, TextResponse), - ({'headers': Headers({'Content-Type': ['application/x-javascript']})}, TextResponse), - ({'headers': Headers({'Content-Type': ['application/json-amazonui-streaming']})}, TextResponse), - ({'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"']}), - 'url': 'http://www.example.com/page/'}, Response), - ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), - ({'url': 'http://www.example.com/page/file.html', - 'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse), - ({'url': 'http://www.example.com/item/file.xml', - 'headers': Headers({'Content-Type': 'application/octet-stream'})}, XmlResponse), - ({'url': 'http://www.example.com/item/file.html', - 'headers': Headers({'Content-Type': 'application/octet-stream'})}, HtmlResponse), - ({'url': 'http://www.example.com/item/file.xml', - 'headers': Headers({'Content-Disposition': ['attachment; filename="data.xml.gz"'], - 'Content-Type': 'application/octet-stream'})}, XmlResponse), - ({'url': 'http://www.example.com/item/file.pdf'}, Response), - ({'filename': 'file.pdf'}, Response), - ({'body': b'\n.', 'url': 'http://www.example.com'}, HtmlResponse), - ] - for source, cls in mappings: - old_cls = _PRE_XTRACTMIME_RESPONSE_TYPES.from_args(**source) - new_cls = responsetypes.from_args(**source) - message = f"{source} ==> {old_cls} (old) != {new_cls} (current)" - assert old_cls == new_cls, message - assert new_cls is cls, f"{source} ==> {new_cls} != {cls}" - - def test_from_args_post_xtractmime(self): - """Each of the following test cases got affected after - using xtractmime for MIME sniffing""" - mappings = [ - ({'body': b'\x00\x01\xff', 'url': 'http://www.example.com/item/', - 'headers': Headers({'Content-Type': ['text/plain']})}, Response), - ({'filename': '/tmp/temp^'}, TextResponse), - ({'body': b'%PDF-1.4'}, Response), - ({'headers': Headers({'Content-Type': ['application/ecmascript']})}, TextResponse), - ({'headers': Headers({'Content-Type': ['application/ld+json']})}, TextResponse), - ({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/html']})}, - HtmlResponse), - ({'headers': Headers({'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']})}, - TextResponse), - ({'body': b'Non HTML', 'headers': Headers({'Content-Encoding': ['zip'], - 'Content-Type': ['text/html']})}, HtmlResponse), - ({'body': b'Some plain text', 'headers': Headers({'Content-Type': 'application/octet-stream'})}, - Response), - ({'body': b'\x0c\x1b'}, TextResponse), - ({'body': b'this is not '}, TextResponse), - ({'body': b'this is not {old_cls} (old) == {new_cls} (current)" - assert old_cls != new_cls, message - assert new_cls is cls, f"{source} ==> {new_cls} != {cls}" - def test_custom_mime_types_loaded(self): - """Check that mime.types files shipped with Scrapy are loaded.""" - self.assertEqual( - _MIME_TYPES.guess_type('x.scrapytest')[0], - 'x-scrapy/test', - ) + # check that mime.types files shipped with scrapy are loaded + self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test') if __name__ == "__main__": diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 0a09f6109..f67fdb40e 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -2,15 +2,235 @@ import os import unittest from urllib.parse import urlparse -from scrapy.http import Response, TextResponse, HtmlResponse +import pytest + +from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse +from scrapy.http.headers import Headers from scrapy.utils.python import to_bytes -from scrapy.utils.response import (response_httprepr, open_in_browser, - get_meta_refresh, get_base_url, response_status_message) +from scrapy.utils.response import ( + get_meta_refresh, + get_response_class, + open_in_browser, + response_httprepr, + response_status_message, +) __doctests__ = ['scrapy.utils.response'] +# Scenarios that work the same with the previously-used, deprecated +# scrapy.responsetypes.responsetypes.from_args +PRE_XTRACTMIME_SCENARIOS = ( + ( + { + 'url': 'http://www.example.com/data.csv', + }, + TextResponse, + ), + ( + { + 'url': 'http://www.example.com/item/', + 'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), + }, + HtmlResponse, + ), + ( + { + 'url': 'http://www.example.com/page/', + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="data.xml.gz"', + ] + } + ), + 'body': b'\x01\x02', + }, + Response, + ), + ( + {'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, + TextResponse, + ), + ( + { + 'body': b'\x03\x02\xdf\xdd\x23', + 'headers': Headers({'Content-Encoding': 'UTF-8'}), + }, + Response, + ), + ( + { + 'body': b'\x00\x01\xff', + 'url': '://www.example.com/item/', + 'headers': Headers({'Content-Type': ['text/plain']}), + }, + TextResponse, + ), + ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), + ({'body': b'Hello'}, HtmlResponse), + ({'body': b'\n.'}, HtmlResponse), + ( + { + 'body': b'\x01\x02', + 'headers': Headers({'Content-Type': ['application/pdf']}), + }, + Response, + ), + ( + {'headers': Headers({'Content-Type': ['application/x-json']})}, + TextResponse, + ), + ( + {'headers': Headers({'Content-Type': ['application/x-javascript']})}, + TextResponse, + ), + ( + { + 'headers': Headers( + {'Content-Type': ['application/json-amazonui-streaming']} + ) + }, + TextResponse, + ), + ( + { + 'headers': Headers( + {'Content-Disposition': ['attachment; filename="data.xml.gz"']} + ), + 'url': 'http://www.example.com/page/', + }, + Response, + ), + ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), + ( + { + 'url': 'http://www.example.com/page/file.html', + 'headers': Headers({'Content-Type': 'application/octet-stream'}), + }, + HtmlResponse, + ), + ( + { + 'url': 'http://www.example.com/item/file.xml', + 'headers': Headers({'Content-Type': 'application/octet-stream'}), + }, + XmlResponse, + ), + ( + { + 'url': 'http://www.example.com/item/file.html', + 'headers': Headers({'Content-Type': 'application/octet-stream'}), + }, + HtmlResponse, + ), + ( + { + 'url': 'http://www.example.com/item/file.xml', + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="data.xml.gz"' + ], + 'Content-Type': 'application/octet-stream', + } + ), + }, + XmlResponse, + ), + ({'url': 'http://www.example.com/item/file.pdf'}, Response), + ({'filename': 'file.pdf'}, Response), + ( + { + 'body': b'\n.', + 'url': 'http://www.example.com', + }, + HtmlResponse, + ), +) + +# Scenarios that work differently with the previously-used, deprecated +# scrapy.responsetypes.responsetypes.from_args +POST_XTRACTMIME_SCENARIOS = ( + ( + { + 'body': b'\x00\x01\xff', + 'url': 'http://www.example.com/item/', + 'headers': Headers({'Content-Type': ['text/plain']}), + }, + Response, + ), + ({'filename': '/tmp/temp^'}, TextResponse), + ({'body': b'%PDF-1.4'}, Response), + ( + {'headers': Headers({'Content-Type': ['application/ecmascript']})}, + TextResponse, + ), + ( + {'headers': Headers({'Content-Type': ['application/ld+json']})}, + TextResponse, + ), + ( + { + 'headers': Headers( + {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} + ) + }, + HtmlResponse, + ), + ( + { + 'headers': Headers( + {'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']} + ) + }, + TextResponse, + ), + ( + { + 'body': b'Non HTML', + 'headers': Headers( + {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} + ), + }, + HtmlResponse, + ), + ( + { + 'body': b'Some plain text', + 'headers': Headers({'Content-Type': 'application/octet-stream'}), + }, + Response, + ), + ({'body': b'\x0c\x1b'}, TextResponse), + ({'body': b'this is not '}, TextResponse), + ({'body': b'this is not - - blahablsdfsal& - """) - self.assertEqual(get_base_url(resp), "http://www.example.com/img/") - - resp2 = HtmlResponse("http://www.example.com", body=b""" - blahablsdfsal&""") - self.assertEqual(get_base_url(resp2), "http://www.example.com") - def test_response_status_message(self): self.assertEqual(response_status_message(200), '200 OK') self.assertEqual(response_status_message(404), '404 Not Found') diff --git a/tox.ini b/tox.ini index a65395a6e..dba61590a 100644 --- a/tox.ini +++ b/tox.ini @@ -63,7 +63,8 @@ commands = pytest --flake8 {posargs:docs scrapy tests} [testenv:pylint] -basepython = python3 +# extra deps require Python 3.8 or lower +basepython = python3.8 deps = {[testenv:extra-deps]deps} pylint==2.12.2 @@ -118,6 +119,8 @@ setenv = {[pinned]setenv} [testenv:extra-deps] +# reppy requires Python 3.8 or lower +basepython = python3.8 deps = {[testenv]deps} boto From 1ea5a8e9e0f9922218cba77724744af85eb321b6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 15:29:33 +0200 Subject: [PATCH 33/81] Deprecate ResponseTypes itself --- scrapy/responsetypes.py | 17 +++++++++++++++-- tests/test_responsetypes.py | 13 ++++++++++++- 2 files changed, 27 insertions(+), 3 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 7308e0350..180cb094e 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -2,7 +2,7 @@ This module implements a class which returns the appropriate Response class based on different criteria. """ -from warnings import warn +from warnings import catch_warnings, simplefilter, warn from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response @@ -30,6 +30,17 @@ class ResponseTypes: 'text/*': 'scrapy.http.TextResponse', } + def __new__(cls, *args, **kwargs): + warn( + ( + 'scrapy.responsetypes.ResponseTypes is deprecated, use ' + 'scrapy.utils.response.get_response_class instead' + ), + ScrapyDeprecationWarning, + stacklevel=2, + ) + return super().__new__(cls) + def __init__(self): self.classes = {} self.mimetypes = _MIME_TYPES @@ -131,4 +142,6 @@ class ResponseTypes: return cls -responsetypes = ResponseTypes() +with catch_warnings(): + simplefilter("ignore") + responsetypes = ResponseTypes() diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 3d5217385..4ce41f086 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -1,4 +1,5 @@ import unittest +from warnings import catch_warnings import pytest @@ -9,7 +10,7 @@ from scrapy.http import ( TextResponse, XmlResponse, ) -from scrapy.responsetypes import responsetypes +from scrapy.responsetypes import responsetypes, ResponseTypes from .test_utils_response import ( POST_XTRACTMIME_SCENARIOS, PRE_XTRACTMIME_SCENARIOS, @@ -116,6 +117,16 @@ class ResponseTypesTest(unittest.TestCase): # check that mime.types files shipped with scrapy are loaded self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test') + def test_class_deprecation(self): + with catch_warnings(record=True) as warnings: + ResponseTypes() + expected_message = ( + 'scrapy.responsetypes.ResponseTypes is deprecated, use ' + 'scrapy.utils.response.get_response_class instead' + ) + messages = {str(warning.message) for warning in warnings} + self.assertIn(expected_message, messages) + if __name__ == "__main__": unittest.main() From 54a024830182d2f799a36789c5a8cf4cf18fc6aa Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 16:24:42 +0200 Subject: [PATCH 34/81] Deprecate ResponseTypes.CLASSES --- scrapy/responsetypes.py | 23 ++++++++++++++++++++++- tests/test_responsetypes.py | 18 ++++++++++++++++++ 2 files changed, 40 insertions(+), 1 deletion(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 180cb094e..46dc28efb 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -11,7 +11,19 @@ from scrapy.utils.python import binary_is_text, to_bytes, to_unicode from scrapy.utils.response import _MIME_TYPES -class ResponseTypes: +class _ResponseTypesMeta(type): + + def __getattribute__(self, name): + if name == 'CLASSES': + warn( + 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated', + ScrapyDeprecationWarning, + stacklevel=2, + ) + return type.__getattribute__(self, name) + + +class ResponseTypes(metaclass=_ResponseTypesMeta): CLASSES = { 'text/html': 'scrapy.http.HtmlResponse', @@ -47,6 +59,15 @@ class ResponseTypes: for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) + def __getattribute__(self, name): + if name == 'CLASSES': + warn( + 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated', + ScrapyDeprecationWarning, + stacklevel=2, + ) + return super().__getattribute__(name) + def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" warn('ResponseTypes.from_mimetype is deprecated, ' diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 4ce41f086..434993c0b 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -127,6 +127,24 @@ class ResponseTypesTest(unittest.TestCase): messages = {str(warning.message) for warning in warnings} self.assertIn(expected_message, messages) + def test_class_classes_deprecation(self): + with catch_warnings(record=True) as warnings: + ResponseTypes.CLASSES + expected_message = ( + 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated' + ) + messages = {str(warning.message) for warning in warnings} + self.assertIn(expected_message, messages) + + def test_instance_classes_deprecation(self): + with catch_warnings(record=True) as warnings: + responsetypes.CLASSES + expected_message = ( + 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated' + ) + messages = {str(warning.message) for warning in warnings} + self.assertIn(expected_message, messages) + if __name__ == "__main__": unittest.main() From 55105ddc7cc61c51cb7f1b8997a3cd4afef459c5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 16:29:18 +0200 Subject: [PATCH 35/81] Do not warn about internal usage of ResponseTypes.CLASSES --- scrapy/responsetypes.py | 5 ++++- tests/test_responsetypes.py | 4 ++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 46dc28efb..6ae6dbd9a 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -56,7 +56,10 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): def __init__(self): self.classes = {} self.mimetypes = _MIME_TYPES - for mimetype, cls in self.CLASSES.items(): + with catch_warnings(): + simplefilter("ignore") + classes_items = self.CLASSES.items() + for mimetype, cls in classes_items: self.classes[mimetype] = load_object(cls) def __getattribute__(self, name): diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 434993c0b..5f131bd07 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -126,6 +126,10 @@ class ResponseTypesTest(unittest.TestCase): ) messages = {str(warning.message) for warning in warnings} self.assertIn(expected_message, messages) + unxepected_message = ( + 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated' + ) + self.assertNotIn(unxepected_message, messages) def test_class_classes_deprecation(self): with catch_warnings(record=True) as warnings: From 948b9ba12567bb9b7ff5a76aa9c173acc1589aa5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 16:36:08 +0200 Subject: [PATCH 36/81] Deprecate scrapy.responsetypes --- scrapy/responsetypes.py | 65 +++++++------------------------------ tests/test_responsetypes.py | 32 ------------------ 2 files changed, 12 insertions(+), 85 deletions(-) diff --git a/scrapy/responsetypes.py b/scrapy/responsetypes.py index 6ae6dbd9a..3eeb9cae5 100644 --- a/scrapy/responsetypes.py +++ b/scrapy/responsetypes.py @@ -2,7 +2,7 @@ This module implements a class which returns the appropriate Response class based on different criteria. """ -from warnings import catch_warnings, simplefilter, warn +from warnings import warn from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import Response @@ -11,19 +11,17 @@ from scrapy.utils.python import binary_is_text, to_bytes, to_unicode from scrapy.utils.response import _MIME_TYPES -class _ResponseTypesMeta(type): - - def __getattribute__(self, name): - if name == 'CLASSES': - warn( - 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated', - ScrapyDeprecationWarning, - stacklevel=2, - ) - return type.__getattribute__(self, name) +warn( + ( + 'scrapy.responsetypes is deprecated, use ' + 'scrapy.utils.response.get_response_class instead' + ), + ScrapyDeprecationWarning, + stacklevel=2, +) -class ResponseTypes(metaclass=_ResponseTypesMeta): +class ResponseTypes: CLASSES = { 'text/html': 'scrapy.http.HtmlResponse', @@ -42,39 +40,14 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): 'text/*': 'scrapy.http.TextResponse', } - def __new__(cls, *args, **kwargs): - warn( - ( - 'scrapy.responsetypes.ResponseTypes is deprecated, use ' - 'scrapy.utils.response.get_response_class instead' - ), - ScrapyDeprecationWarning, - stacklevel=2, - ) - return super().__new__(cls) - def __init__(self): self.classes = {} self.mimetypes = _MIME_TYPES - with catch_warnings(): - simplefilter("ignore") - classes_items = self.CLASSES.items() - for mimetype, cls in classes_items: + for mimetype, cls in self.CLASSES.items(): self.classes[mimetype] = load_object(cls) - def __getattribute__(self, name): - if name == 'CLASSES': - warn( - 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated', - ScrapyDeprecationWarning, - stacklevel=2, - ) - return super().__getattribute__(name) - def from_mimetype(self, mimetype): """Return the most appropriate Response class for the given mimetype""" - warn('ResponseTypes.from_mimetype is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) if mimetype is None: return Response elif mimetype in self.classes: @@ -86,16 +59,12 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): def from_content_type(self, content_type, content_encoding=None): """Return the most appropriate Response class from an HTTP Content-Type header """ - warn('ResponseTypes.from_content_type is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) if content_encoding: return Response mimetype = to_unicode(content_type).split(';')[0].strip().lower() return self.from_mimetype(mimetype) def from_content_disposition(self, content_disposition): - warn('ResponseTypes.from_content_disposition is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) try: filename = to_unicode( content_disposition, encoding='latin-1', errors='replace' @@ -107,8 +76,6 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): def from_headers(self, headers): """Return the most appropriate Response class by looking at the HTTP headers""" - warn('ResponseTypes.from_headers is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) cls = Response if b'Content-Type' in headers: cls = self.from_content_type( @@ -121,8 +88,6 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): def from_filename(self, filename): """Return the most appropriate Response class from a file name""" - warn('ResponseTypes.from_filename is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) mimetype, encoding = self.mimetypes.guess_type(filename) if mimetype and not encoding: return self.from_mimetype(mimetype) @@ -134,8 +99,6 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): This method is a bit magic and could be improved in the future, but it's not meant to be used except for special cases where response types cannot be guess using more straightforward methods.""" - warn('ResponseTypes.from_body is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) chunk = body[:5000] chunk = to_bytes(chunk) if not binary_is_text(chunk): @@ -152,8 +115,6 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): def from_args(self, headers=None, url=None, filename=None, body=None): """Guess the most appropriate Response class based on the given arguments.""" - warn('ResponseTypes.from_args is deprecated, ' - 'please use scrapy.utils.response.get_response_class instead', ScrapyDeprecationWarning) cls = Response if headers is not None: cls = self.from_headers(headers) @@ -166,6 +127,4 @@ class ResponseTypes(metaclass=_ResponseTypesMeta): return cls -with catch_warnings(): - simplefilter("ignore") - responsetypes = ResponseTypes() +responsetypes = ResponseTypes() diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 5f131bd07..0b6be7f85 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -117,38 +117,6 @@ class ResponseTypesTest(unittest.TestCase): # check that mime.types files shipped with scrapy are loaded self.assertEqual(responsetypes.mimetypes.guess_type('x.scrapytest')[0], 'x-scrapy/test') - def test_class_deprecation(self): - with catch_warnings(record=True) as warnings: - ResponseTypes() - expected_message = ( - 'scrapy.responsetypes.ResponseTypes is deprecated, use ' - 'scrapy.utils.response.get_response_class instead' - ) - messages = {str(warning.message) for warning in warnings} - self.assertIn(expected_message, messages) - unxepected_message = ( - 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated' - ) - self.assertNotIn(unxepected_message, messages) - - def test_class_classes_deprecation(self): - with catch_warnings(record=True) as warnings: - ResponseTypes.CLASSES - expected_message = ( - 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated' - ) - messages = {str(warning.message) for warning in warnings} - self.assertIn(expected_message, messages) - - def test_instance_classes_deprecation(self): - with catch_warnings(record=True) as warnings: - responsetypes.CLASSES - expected_message = ( - 'scrapy.responsetypes.ResponseTypes.CLASSES is deprecated' - ) - messages = {str(warning.message) for warning in warnings} - self.assertIn(expected_message, messages) - if __name__ == "__main__": unittest.main() From 983aa66687390abe4aad89df1263f7a02f5766d5 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 16:49:13 +0200 Subject: [PATCH 37/81] Test ResponseTypes MIME types with the new implementation --- scrapy/utils/response.py | 14 ++++++++++++-- tests/test_utils_response.py | 13 +++++++++++++ 2 files changed, 25 insertions(+), 2 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 6d1b575b8..fd0b59de4 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -61,6 +61,16 @@ _mime_overrides = get_data('scrapy', 'mime.types') or b'' _MIME_TYPES.readfp(StringIO(_mime_overrides.decode())) + +def _is_html_mime_type(mime_type): + if mime_type in { + b'application/xhtml+xml', + b'application/vnd.wap.xhtml+xml', + }: + return True + return is_html_mime_type(mime_type) + + def _is_other_text_mime_type(mime_type): return ( mime_type.startswith(b'text/') @@ -75,7 +85,7 @@ def _is_other_text_mime_type(mime_type): _PRIORITIZED_MIME_TYPE_CHECKERS = ( - is_html_mime_type, + _is_html_mime_type, is_xml_mime_type, _is_other_text_mime_type, ) @@ -123,7 +133,7 @@ def _get_mime_type_from_path(path): def _get_response_class_from_mime_type(mime_type): if not mime_type: return Response - if is_html_mime_type(mime_type): + if _is_html_mime_type(mime_type): return HtmlResponse if is_xml_mime_type(mime_type): return XmlResponse diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index f67fdb40e..bd229881f 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -6,6 +6,8 @@ import pytest from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse from scrapy.http.headers import Headers +from scrapy.responsetypes import ResponseTypes +from scrapy.utils.misc import load_object from scrapy.utils.python import to_bytes from scrapy.utils.response import ( get_meta_refresh, @@ -22,6 +24,17 @@ __doctests__ = ['scrapy.utils.response'] # Scenarios that work the same with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args PRE_XTRACTMIME_SCENARIOS = ( + *( + ( + { + 'headers': Headers( + {'Content-Type': [mime_type]} + ), + }, + load_object(class_path), + ) + for mime_type, class_path in ResponseTypes.CLASSES.items() + ), ( { 'url': 'http://www.example.com/data.csv', From 4759c681934876351aea5ca18175e4f4efcc8633 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 16:57:17 +0200 Subject: [PATCH 38/81] Address style issues --- scrapy/utils/response.py | 1 - tests/test_responsetypes.py | 3 +-- 2 files changed, 1 insertion(+), 3 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index fd0b59de4..4acc77111 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -61,7 +61,6 @@ _mime_overrides = get_data('scrapy', 'mime.types') or b'' _MIME_TYPES.readfp(StringIO(_mime_overrides.decode())) - def _is_html_mime_type(mime_type): if mime_type in { b'application/xhtml+xml', diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 0b6be7f85..3d5217385 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -1,5 +1,4 @@ import unittest -from warnings import catch_warnings import pytest @@ -10,7 +9,7 @@ from scrapy.http import ( TextResponse, XmlResponse, ) -from scrapy.responsetypes import responsetypes, ResponseTypes +from scrapy.responsetypes import responsetypes from .test_utils_response import ( POST_XTRACTMIME_SCENARIOS, PRE_XTRACTMIME_SCENARIOS, From 7ef061fc173ed7cedb5baa14e08d2ba38fbc06c2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 17:10:59 +0200 Subject: [PATCH 39/81] HttpCompressionMiddleware: clarify internal comment --- scrapy/downloadermiddlewares/httpcompression.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 11407ca46..a3e5b3526 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -70,8 +70,8 @@ class HttpCompressionMiddleware: ) kwargs = dict(cls=respcls, body=decoded_body) if issubclass(respcls, TextResponse): - # force recalculating the encoding until we make sure the - # responsetypes guessing is reliable + # Force recalculating the encoding based on the new, + # decoded (uncompressed) body. kwargs['encoding'] = None response = response.replace(**kwargs) if not content_encoding: From 3d7e2e5021c74f0384726d65ce4e0c0dc55ef05a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 17:42:51 +0200 Subject: [PATCH 40/81] Add a test for the data URI handler now taking the body into account --- tests/test_downloader_handlers.py | 13 +++++++++++++ 1 file changed, 13 insertions(+) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index e9f890391..bbf641439 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -3,6 +3,7 @@ import os import shutil import sys import tempfile +from base64 import b64encode from typing import Optional, Type from unittest import mock @@ -1237,3 +1238,15 @@ class DataURITestCase(unittest.TestCase): request = Request("data:,") return self.download_request(request, self.spider).addCallback(_test) + + def test_body_mime_type(self): + """Test that the body, and not only the declared MIME type, is taken + into account when choosing a response class.""" + def _test(response): + self.assertIsInstance(response, HtmlResponse) + + html = '\n.' + base64_html = b64encode(html.encode()).decode() + data_uri = f'data:application/unknown;base64,{base64_html}' + request = Request(data_uri) + return self.download_request(request, self.spider).addCallback(_test) From b7e88f4357eb11274693ece3618fb92cf7052d12 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 19:12:13 +0200 Subject: [PATCH 41/81] test_get_response_class_http: cover Apache bug scenarios --- tests/test_responsetypes.py | 9 +++++- tests/test_utils_response.py | 61 +++++++++++++++++++++++++++++++----- 2 files changed, 62 insertions(+), 8 deletions(-) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 3d5217385..42d4e481d 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -16,10 +16,17 @@ from .test_utils_response import ( ) +def _unmark(item): + return pytest.param(*item.values) + + @pytest.mark.parametrize( "kwargs,response_class", ( - *PRE_XTRACTMIME_SCENARIOS, + *( + item if not hasattr(item, "marks") else _unmark(item) + for item in PRE_XTRACTMIME_SCENARIOS + ), *( pytest.param( kwargs, diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index bd229881f..dbdb57b1b 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -24,6 +24,36 @@ __doctests__ = ['scrapy.utils.response'] # Scenarios that work the same with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args PRE_XTRACTMIME_SCENARIOS = ( + # Even if the body is binary, if the Content-Type says it is text, we + # interpret it as text, as long as the Content-Type is not one of the 4 + # affected by the Apache bug. + # + # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata + ( + { + 'body': b'\x00\x01\xff', + 'headers': Headers({'Content-Type': ['text/json']}), + }, + TextResponse, + ), + *( + pytest.param( + { + 'body': b'\x00\x01\xff', + 'headers': Headers({'Content-Type': [content_type]}), + }, + TextResponse, + marks=pytest.mark.xfail( + strict=True, + reason="https://github.com/scrapy/xtractmime/issues/13", + ), + ) + for content_type in ( + 'text/plain; charset=Iso-8859-1', + 'text/plain; charset=utf-8', + 'text/plain; charset=windows-1252', + ) + ), *( ( { @@ -172,14 +202,30 @@ PRE_XTRACTMIME_SCENARIOS = ( # Scenarios that work differently with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args POST_XTRACTMIME_SCENARIOS = ( - ( - { - 'body': b'\x00\x01\xff', - 'url': 'http://www.example.com/item/', - 'headers': Headers({'Content-Type': ['text/plain']}), - }, - Response, + # A known Apache bug may cause a server to send files with Content-Type set + # to "text/plain", "text/plain; charset=ISO-8859-1", + # "text/plain; charset=iso-8859-1", or "text/plain; charset=UTF-8", + # regardless of the actual file content. + # + # They should be treated as binary if their content is binary. + # + # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata + *( + ( + { + 'body': b'\x00\x01\xff', + 'headers': Headers({'Content-Type': [content_type]}), + }, + Response, + ) + for content_type in ( + 'text/plain', + 'text/plain; charset=ISO-8859-1', + 'text/plain; charset=iso-8859-1', + 'text/plain; charset=UTF-8', + ) ), + ({'filename': '/tmp/temp^'}, TextResponse), ({'body': b'%PDF-1.4'}, Response), ( @@ -236,6 +282,7 @@ POST_XTRACTMIME_SCENARIOS = ( ), ) def test_get_response_class_http(kwargs, response_class): + kwargs = dict(kwargs) if 'headers' in kwargs: kwargs['http_headers'] = kwargs.pop('headers') if 'filename' in kwargs: From 0a77bfe2a0086abffe2681fdce659d9115712433 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Thu, 30 Jun 2022 20:39:34 +0200 Subject: [PATCH 42/81] get_response_class: use Response for compressed data --- scrapy/utils/response.py | 62 ++++++++++---- tests/test_utils_response.py | 154 ++++++++++++++++++++++++++--------- 2 files changed, 159 insertions(+), 57 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 4acc77111..825272274 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -51,10 +51,10 @@ from scrapy.utils.python import to_bytes, to_unicode _baseurl_cache: "WeakKeyDictionary[Response, str]" = WeakKeyDictionary() -_CONTENT_ENCODING_MIME_TYPES = { - 'br': b'application/brotli', - 'deflate': b'application/zip', - 'gzip': b'application/gzip', +_ENCODING_MIME_TYPES = { + b'br': b'application/brotli', + b'compress': b'application/x-compress', + b'deflate': b'application/zip', } _MIME_TYPES = MimeTypes() _mime_overrides = get_data('scrapy', 'mime.types') or b'' @@ -103,10 +103,12 @@ def _get_best_mime_type(mime_types): return mime_types[0] -def _get_http_header_mime_types(headers: Headers) -> Sequence[bytes]: +def _get_encoding_or_mime_types_from_headers( + headers: Headers, +) -> Sequence[bytes]: mime_types = [] - if b'Content-Type' in headers: - mime_types.append(headers[b'Content-Type'].split(b';')[0]) + if b'Content-Encoding' in headers: + return headers.getlist(b'Content-Encoding')[-1], None if b'Content-Disposition' in headers: path = ( headers.get(b"Content-Disposition") @@ -115,18 +117,29 @@ def _get_http_header_mime_types(headers: Headers) -> Sequence[bytes]: .strip(b"\"'") .decode() ) - mime_types.append(_get_mime_type_from_path(path)) - return mime_types + encoding, mime_type = _get_encoding_or_mime_type_from_path(path) + if encoding: + return encoding, None + mime_types.append(mime_type) + if b'Content-Type' in headers: + mime_types.append(headers[b'Content-Type'].split(b';')[0]) + return None, mime_types -def _get_mime_type_from_path(path): +def _get_mime_type_from_encoding(encoding): + return ( + _ENCODING_MIME_TYPES.get(encoding, None) + or b"application/" + encoding + ) + + +def _get_encoding_or_mime_type_from_path(path): mimetype, encoding = _MIME_TYPES.guess_type(path, strict=False) - encoding_mime_type = _CONTENT_ENCODING_MIME_TYPES.get(encoding, None) - if encoding_mime_type: - return encoding_mime_type + if encoding: + return encoding.encode(), None if mimetype: - return mimetype.encode() - return None + return None, mimetype.encode() + return None, None def _get_response_class_from_mime_type(mime_type): @@ -195,16 +208,29 @@ def get_response_class( """Guess the most appropriate Response class based on the given arguments.""" mime_types = list(declared_mime_types or []) + encoding = None # as in compression (e.g. gzip), not charset if http_headers: - mime_types.extend(_get_http_header_mime_types(http_headers)) + encoding, header_mime_types = ( + _get_encoding_or_mime_types_from_headers(http_headers) + ) + if not encoding: + mime_types.extend(header_mime_types) if url is not None: url_parts = urlparse(url) http_origin = url_parts.scheme in ("http", "https") - mime_types.append(_get_mime_type_from_path(url_parts.path)) + if not encoding: + encoding, path_mime_type = ( + _get_encoding_or_mime_type_from_path(url_parts.path) + ) + if not encoding: + mime_types.append(path_mime_type) else: http_origin = True body = _remove_nul_byte_from_text((body or b'')[:BODY_LIMIT]) - if mime_types: + if encoding: + best_mime_type = _get_mime_type_from_encoding(encoding) + content_types = (best_mime_type,) + elif mime_types: best_mime_type = _get_best_mime_type(mime_types) content_types = (best_mime_type,) if best_mime_type else best_mime_type else: diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index dbdb57b1b..91a1e5457 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -54,6 +54,67 @@ PRE_XTRACTMIME_SCENARIOS = ( 'text/plain; charset=windows-1252', ) ), + + # JavaScript MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#javascript-mime-type + *( + ( + {'headers': Headers({'Content-Type': [content_type]})}, + TextResponse, + ) + for content_type in ( + 'application/javascript', + 'application/x-javascript', + 'text/ecmascript', + 'text/javascript', + 'text/javascript1.0', + 'text/javascript1.1', + 'text/javascript1.2', + 'text/javascript1.3', + 'text/javascript1.4', + 'text/javascript1.5', + 'text/jscript', + 'text/livescript', + 'text/x-ecmascript', + 'text/x-javascript', + ) + ), + + # JSON MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#json-mime-type + *( + ( + {'headers': Headers({'Content-Type': [content_type]})}, + TextResponse, + ) + for content_type in ( + 'application/json', + 'text/json', + ) + ), + + # Compressed content should be of type Response until uncompressed. + *( + ( + { + 'headers': Headers( + { + 'Content-Encoding': ['zip'], + 'Content-Type': [content_type], + } + ) + }, + Response, + ) + for content_type in ( + 'text/html', + 'text/xml', + 'text/plain', + ) + ), + *( ( { @@ -174,20 +235,6 @@ PRE_XTRACTMIME_SCENARIOS = ( }, HtmlResponse, ), - ( - { - 'url': 'http://www.example.com/item/file.xml', - 'headers': Headers( - { - 'Content-Disposition': [ - 'attachment; filename="data.xml.gz"' - ], - 'Content-Type': 'application/octet-stream', - } - ), - }, - XmlResponse, - ), ({'url': 'http://www.example.com/item/file.pdf'}, Response), ({'filename': 'file.pdf'}, Response), ( @@ -207,7 +254,8 @@ POST_XTRACTMIME_SCENARIOS = ( # "text/plain; charset=iso-8859-1", or "text/plain; charset=UTF-8", # regardless of the actual file content. # - # They should be treated as binary if their content is binary. + # They should be treated as binary if their content is binary, and as + # text/plain otherwise. # # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata *( @@ -226,32 +274,46 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - ({'filename': '/tmp/temp^'}, TextResponse), + # If the body is empty, it contains no binary data bytes, hence body-based + # MIME type detection must interpret the result as text. + # + # https://mimesniff.spec.whatwg.org/#identifying-a-resource-with-an-unknown-mime-type + ({}, TextResponse), + ({'url': '/tmp/temp^'}, TextResponse), + + # Body-based PDF detection + # + # https://mimesniff.spec.whatwg.org/#identifying-a-resource-with-an-unknown-mime-type ({'body': b'%PDF-1.4'}, Response), - ( - {'headers': Headers({'Content-Type': ['application/ecmascript']})}, - TextResponse, + + # JavaScript MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#javascript-mime-type + *( + ( + {'headers': Headers({'Content-Type': [content_type]})}, + TextResponse, + ) + for content_type in ( + 'application/ecmascript', + 'application/x-ecmascript', + ) ), - ( - {'headers': Headers({'Content-Type': ['application/ld+json']})}, - TextResponse, - ), - ( - { - 'headers': Headers( - {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} - ) - }, - HtmlResponse, - ), - ( - { - 'headers': Headers( - {'Content-Encoding': ['zip'], 'Content-Type': ['text/plain']} - ) - }, - TextResponse, + + # JSON MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#json-mime-type + *( + ( + {'headers': Headers({'Content-Type': [content_type]})}, + TextResponse, + ) + for content_type in ( + 'application/foo+json', + 'application/ld+json', + ) ), + ( { 'body': b'Non HTML', @@ -259,7 +321,7 @@ POST_XTRACTMIME_SCENARIOS = ( {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} ), }, - HtmlResponse, + Response, ), ( { @@ -271,6 +333,20 @@ POST_XTRACTMIME_SCENARIOS = ( ({'body': b'\x0c\x1b'}, TextResponse), ({'body': b'this is not '}, TextResponse), ({'body': b'this is not Date: Thu, 30 Jun 2022 21:28:02 +0200 Subject: [PATCH 43/81] test_get_response_class_http: progress on documenting behavior changes --- scrapy/utils/response.py | 16 ++++++-- tests/test_utils_response.py | 75 +++++++++++++++++++++++++++++++----- 2 files changed, 79 insertions(+), 12 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 825272274..bf479eeb3 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -51,16 +51,23 @@ from scrapy.utils.python import to_bytes, to_unicode _baseurl_cache: "WeakKeyDictionary[Response, str]" = WeakKeyDictionary() -_ENCODING_MIME_TYPES = { +_ENCODING_MIME_TYPE_MAP = { b'br': b'application/brotli', b'compress': b'application/x-compress', b'deflate': b'application/zip', + b'gzip': b'application/gzip', + b'zstd': b'application/zstd', } +_ENCODING_MIME_TYPES = {*_ENCODING_MIME_TYPE_MAP.values()} _MIME_TYPES = MimeTypes() _mime_overrides = get_data('scrapy', 'mime.types') or b'' _MIME_TYPES.readfp(StringIO(_mime_overrides.decode())) +def _is_compressed_mime_type(mime_type): + return mime_type in _ENCODING_MIME_TYPES + + def _is_html_mime_type(mime_type): if mime_type in { b'application/xhtml+xml', @@ -84,6 +91,7 @@ def _is_other_text_mime_type(mime_type): _PRIORITIZED_MIME_TYPE_CHECKERS = ( + _is_compressed_mime_type, _is_html_mime_type, is_xml_mime_type, _is_other_text_mime_type, @@ -108,7 +116,9 @@ def _get_encoding_or_mime_types_from_headers( ) -> Sequence[bytes]: mime_types = [] if b'Content-Encoding' in headers: - return headers.getlist(b'Content-Encoding')[-1], None + encodings = headers.getlist(b'Content-Encoding') + if encodings: + return encodings[-1], None if b'Content-Disposition' in headers: path = ( headers.get(b"Content-Disposition") @@ -128,7 +138,7 @@ def _get_encoding_or_mime_types_from_headers( def _get_mime_type_from_encoding(encoding): return ( - _ENCODING_MIME_TYPES.get(encoding, None) + _ENCODING_MIME_TYPE_MAP.get(encoding, None) or b"application/" + encoding ) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 91a1e5457..81283112e 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -314,15 +314,72 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - ( - { - 'body': b'Non HTML', - 'headers': Headers( - {'Content-Encoding': ['zip'], 'Content-Type': ['text/html']} - ), - }, - Response, - ), + + # Compressed content should be of type Response until uncompressed. + # + # When it comes to compression, we trust the encoding from the following + # sources, in the following order: + # + # 1, Content-Encoding HTTP header + # + # 2. File extension of the file name of the Content-Disposition HTTP + # header. + # + # 3. File extension of the file name if the resource comes through a + # file-based protocol (FTP, local file system). + #( + #{ + #'url': 'file.html', + #'body': b'\n.', + #'headers': Headers( + #{ + #'Content-Disposition': [ + #'attachment; filename="file.html"' + #], + #'Content-Encoding': ['zip'], + #'Content-Type': ['text/html'], + #} + #), + #}, + #Response, + #), + #( + #{ + #'url': 'file.html', + #'body': b'\n.', + #'headers': Headers( + #{ + #'Content-Disposition': [ + #'attachment; filename="file.html.zip"' + #], + #'Content-Type': ['text/html'], + #} + #), + #}, + #Response, + #), + #( + #{ + #'url': 'file.html.zip', + #'body': b'\n.', + #}, + #Response, + #), + + # HTTP headers take priority over local file extensions. + #( + #{ + #'url': 'file.html.zip', + #'body': b'\n.', + #'headers': Headers( + #{ + #'Content-Type': ['text/html'], + #} + #), + #}, + #HtmlResponse, + #), + ( { 'body': b'Some plain text', From e96b07bd4b8d69907407397a97331250132a77d2 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 11:38:46 +0100 Subject: [PATCH 44/81] flake8: ignore failing file for the time being --- tests/test_utils_response.py | 2 ++ 1 file changed, 2 insertions(+) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 5699d9571..cd9d9c36f 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -1,3 +1,5 @@ +# flake8: noqa TODO: remove this line + import unittest import warnings from pathlib import Path From 160096b7cf2944908b4ce42c32522510d8271105 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 11:43:52 +0100 Subject: [PATCH 45/81] =?UTF-8?q?pytest-cov:=203.0.0=20=E2=86=92=204.0.0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Removes 2 warnings when running tests --- tests/requirements.txt | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/requirements.txt b/tests/requirements.txt index d9373dfa8..618949795 100644 --- a/tests/requirements.txt +++ b/tests/requirements.txt @@ -2,7 +2,7 @@ attrs pyftpdlib pytest -pytest-cov==3.0.0 +pytest-cov==4.0.0 pytest-xdist sybil >= 1.3.0 # https://github.com/cjw296/sybil/issues/20#issuecomment-605433422 testfixtures From 112b232023c43c2bc54020dc3bdc69c66d2f1502 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 11:47:38 +0100 Subject: [PATCH 46/81] Silence the scrapy.responsetypes deprecation warning in tests --- pytest.ini | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/pytest.ini b/pytest.ini index f5fbf2529..4398ae2cd 100644 --- a/pytest.ini +++ b/pytest.ini @@ -22,7 +22,8 @@ markers = only_asyncio: marks tests as only enabled when --reactor=asyncio is passed only_not_asyncio: marks tests as only enabled when --reactor=asyncio is not passed filterwarnings = - ignore:scrapy.downloadermiddlewares.decompression is deprecated ignore:Module scrapy.utils.reqser is deprecated + ignore:scrapy.downloadermiddlewares.decompression is deprecated + ignore:scrapy.responsetypes is deprecated ignore:typing.re is deprecated ignore:typing.io is deprecated From 0d86d845e8d28c697f623f54dbdef48fcbde017b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 11:53:36 +0100 Subject: [PATCH 47/81] Silence a test warning caused by pytest-cov --- .coveragerc | 2 ++ 1 file changed, 2 insertions(+) diff --git a/.coveragerc b/.coveragerc index 02acbff8e..bbe5cebc3 100644 --- a/.coveragerc +++ b/.coveragerc @@ -1,5 +1,7 @@ [run] branch = true +; https://github.com/pytest-dev/pytest-cov/issues/369#issuecomment-1053702088 +disable_warnings=include-ignored include = scrapy/* omit = tests/* From f029be3aa7f75a282dfbd429284abcbca3bf2a9d Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 12:55:17 +0100 Subject: [PATCH 48/81] Fix Apache bug handling --- scrapy/utils/response.py | 2 +- setup.py | 4 +++- tests/test_utils_response.py | 15 +++------------ 3 files changed, 7 insertions(+), 14 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 1ae987b7a..ae13b12ba 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -131,7 +131,7 @@ def _get_encoding_or_mime_types_from_headers( return encoding, None mime_types.append(mime_type) if b'Content-Type' in headers: - mime_types.append(headers[b'Content-Type'].split(b';')[0]) + mime_types.append(headers[b'Content-Type']) return None, mime_types diff --git a/setup.py b/setup.py index 2153c03fc..d5ba186a4 100644 --- a/setup.py +++ b/setup.py @@ -34,7 +34,9 @@ install_requires = [ 'packaging', 'tldextract', 'lxml>=4.3.0', - 'xtractmime>=0.1.0', + # TODO: Release a new version of xtractmime and use it as the minimum + # version here. + 'xtractmime @ git+https://github.com/scrapy/xtractmime@c65d09c94836547dd7c6a659fe9740c112253526', ] extras_require = {} cpython_dependencies = [ diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index cd9d9c36f..3c150faab 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -33,26 +33,17 @@ PRE_XTRACTMIME_SCENARIOS = ( # affected by the Apache bug. # # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata - ( - { - 'body': b'\x00\x01\xff', - 'headers': Headers({'Content-Type': ['text/json']}), - }, - TextResponse, - ), *( - pytest.param( + ( { 'body': b'\x00\x01\xff', 'headers': Headers({'Content-Type': [content_type]}), }, TextResponse, - marks=pytest.mark.xfail( - strict=True, - reason="https://github.com/scrapy/xtractmime/issues/13", - ), ) for content_type in ( + 'text/json', + # text/plain variants *not* affected by the Apache bug 'text/plain; charset=Iso-8859-1', 'text/plain; charset=utf-8', 'text/plain; charset=windows-1252', From d319b62d16fe680cd076009d0b5b4454c9b3e42b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 13:01:14 +0100 Subject: [PATCH 49/81] Fix typing issues --- scrapy/utils/response.py | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index ae13b12ba..1db5d2b17 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -112,7 +112,7 @@ def _get_best_mime_type(mime_types): def _get_encoding_or_mime_types_from_headers( headers: Headers, -) -> Sequence[bytes]: +) -> Tuple[Optional[bytes], Optional[Sequence[bytes]]]: mime_types = [] if b'Content-Encoding' in headers: encodings = headers.getlist(b'Content-Encoding') @@ -209,10 +209,10 @@ def get_meta_refresh( def get_response_class( *, - url: str = None, - body: bytes = None, - declared_mime_types: Sequence[bytes] = None, - http_headers: Headers = None, + url: Optional[str] = None, + body: Optional[bytes] = None, + declared_mime_types: Optional[Sequence[bytes]] = None, + http_headers: Optional[Headers] = None, ) -> Type[Response]: """Guess the most appropriate Response class based on the given arguments.""" @@ -223,6 +223,7 @@ def get_response_class( _get_encoding_or_mime_types_from_headers(http_headers) ) if not encoding: + assert header_mime_types is not None mime_types.extend(header_mime_types) if url is not None: url_parts = urlparse(url) From fa3ba512d992346c5675aaaaa57559e3ced76934 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 13:21:36 +0100 Subject: [PATCH 50/81] Uncomment test scenarios --- tests/test_utils_response.py | 104 +++++++++++++++++------------------ 1 file changed, 51 insertions(+), 53 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 3c150faab..7b7819ee2 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -110,6 +110,19 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), + # HTTP headers take priority over local file extensions. + ( + { + 'url': 'file.txt', + 'headers': Headers( + { + 'Content-Type': ['text/html'], + } + ), + }, + HtmlResponse, + ), + *( ( { @@ -309,7 +322,6 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - # Compressed content should be of type Response until uncompressed. # # When it comes to compression, we trust the encoding from the following @@ -322,58 +334,44 @@ POST_XTRACTMIME_SCENARIOS = ( # # 3. File extension of the file name if the resource comes through a # file-based protocol (FTP, local file system). - #( - #{ - #'url': 'file.html', - #'body': b'\n.', - #'headers': Headers( - #{ - #'Content-Disposition': [ - #'attachment; filename="file.html"' - #], - #'Content-Encoding': ['zip'], - #'Content-Type': ['text/html'], - #} - #), - #}, - #Response, - #), - #( - #{ - #'url': 'file.html', - #'body': b'\n.', - #'headers': Headers( - #{ - #'Content-Disposition': [ - #'attachment; filename="file.html.zip"' - #], - #'Content-Type': ['text/html'], - #} - #), - #}, - #Response, - #), - #( - #{ - #'url': 'file.html.zip', - #'body': b'\n.', - #}, - #Response, - #), - - # HTTP headers take priority over local file extensions. - #( - #{ - #'url': 'file.html.zip', - #'body': b'\n.', - #'headers': Headers( - #{ - #'Content-Type': ['text/html'], - #} - #), - #}, - #HtmlResponse, - #), + ( + { + 'url': 'file.html', + 'body': b'\n.', + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="file.html"' + ], + 'Content-Encoding': ['zip'], + 'Content-Type': ['text/html'], + } + ), + }, + Response, + ), + ( + { + 'url': 'file.html', + 'body': b'\n.', + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="file.html.zip"' + ], + 'Content-Type': ['text/html'], + } + ), + }, + Response, + ), + ( + { + 'url': 'file.html.zip', + 'body': b'\n.', + }, + Response, + ), ( { From 1c169895453f2f564bfab061078ebdac2b5de0bf Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 13:23:43 +0100 Subject: [PATCH 51/81] Upgrade xtractmime on pinned CI environments --- tox.ini | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/tox.ini b/tox.ini index 74559a72f..e1c52be61 100644 --- a/tox.ini +++ b/tox.ini @@ -94,7 +94,9 @@ deps = w3lib==1.17.0 zope.interface==5.1.0 lxml==4.3.0 - xtractmime==0.1.0 + # TODO: Release a new version of xtractmime and use it as the pinned + # version here. + xtractmime@git+https://github.com/scrapy/xtractmime@c65d09c94836547dd7c6a659fe9740c112253526 -rtests/requirements.txt # mitmproxy 4.0.4+ requires upgrading some of the pinned dependencies From e7d36a305d43362cd3a7aa6501eefc637c41d74f Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 14:24:54 +0100 Subject: [PATCH 52/81] Ignore file path extensions in HTTP responses --- scrapy/utils/response.py | 2 +- tests/test_utils_response.py | 89 +++++++++++++++++++----------------- 2 files changed, 49 insertions(+), 42 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 1db5d2b17..0f715e26d 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -228,7 +228,7 @@ def get_response_class( if url is not None: url_parts = urlparse(url) http_origin = url_parts.scheme in ("http", "https") - if not encoding: + if not http_origin and not encoding: encoding, path_mime_type = ( _get_encoding_or_mime_type_from_path(url_parts.path) ) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 7b7819ee2..a27e2cc9f 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -28,6 +28,24 @@ __doctests__ = ['scrapy.utils.response'] # Scenarios that work the same with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args PRE_XTRACTMIME_SCENARIOS = ( + # Content-Type determines the type for the HTTP protocol. + *( + ( + { + "url": f"{protocol}://example.com/foo", + "headers": Headers({"Content-Type": content_type}), + }, + response_class, + ) + for protocol in ("http", "https") + for content_type, response_class in ( + ("application/octet-stream", Response), + ("text/plain", TextResponse), + ("text/html", HtmlResponse), + ("text/xml", XmlResponse), + ) + ), + # Even if the body is binary, if the Content-Type says it is text, we # interpret it as text, as long as the Content-Type is not one of the 4 # affected by the Apache bug. @@ -110,19 +128,8 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - # HTTP headers take priority over local file extensions. - ( - { - 'url': 'file.txt', - 'headers': Headers( - { - 'Content-Type': ['text/html'], - } - ), - }, - HtmlResponse, - ), - + # We continue to support the hard-coded MIME-to-Response-class mappings + # from scrapy.responsetypes. *( ( { @@ -134,12 +141,21 @@ PRE_XTRACTMIME_SCENARIOS = ( ) for mime_type, class_path in ResponseTypes.CLASSES.items() ), - ( - { - 'url': 'http://www.example.com/data.csv', - }, - TextResponse, + + # We take the file extension of URL paths into account, except for HTTP + # responses, because “they are unreliable and easily spoofed”. + # + # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata + *( + ( + {'url': f'{protocol}://example.com/a.html'}, + response_class, + ) + for protocol, response_class in ( + *((protocol, HtmlResponse) for protocol in ("file", "ftp")), + ) ), + ( { 'url': 'http://www.example.com/item/', @@ -180,7 +196,6 @@ PRE_XTRACTMIME_SCENARIOS = ( }, TextResponse, ), - ({'url': 'http://www.example.com/item/file.html'}, HtmlResponse), ({'body': b'Hello'}, HtmlResponse), ({'body': b' Date: Sun, 8 Jan 2023 15:46:32 +0100 Subject: [PATCH 53/81] Always prefer Content-Type over Content-Disposition --- scrapy/utils/response.py | 58 +++++---------- tests/test_utils_response.py | 135 ++++++++++++++++++++--------------- 2 files changed, 95 insertions(+), 98 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 0f715e26d..ea6b33f96 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -89,35 +89,15 @@ def _is_other_text_mime_type(mime_type): ) -_PRIORITIZED_MIME_TYPE_CHECKERS = ( - _is_compressed_mime_type, - _is_html_mime_type, - is_xml_mime_type, - _is_other_text_mime_type, -) - - -def _get_best_mime_type(mime_types): - candidate_mime_types = tuple( - mime_type - for mime_type in mime_types - if mime_type is not None - ) - for mime_type_checker in _PRIORITIZED_MIME_TYPE_CHECKERS: - for candidate_mime_type in candidate_mime_types: - if mime_type_checker(candidate_mime_type): - return candidate_mime_type - return mime_types[0] - - -def _get_encoding_or_mime_types_from_headers( +def _get_encoding_or_mime_type_from_headers( headers: Headers, -) -> Tuple[Optional[bytes], Optional[Sequence[bytes]]]: - mime_types = [] +) -> Tuple[Optional[bytes], Optional[bytes]]: if b'Content-Encoding' in headers: encodings = headers.getlist(b'Content-Encoding') if encodings: return encodings[-1], None + if b'Content-Type' in headers: + return None, headers[b'Content-Type'] if b'Content-Disposition' in headers: path = ( headers.get(b"Content-Disposition") @@ -129,10 +109,8 @@ def _get_encoding_or_mime_types_from_headers( encoding, mime_type = _get_encoding_or_mime_type_from_path(path) if encoding: return encoding, None - mime_types.append(mime_type) - if b'Content-Type' in headers: - mime_types.append(headers[b'Content-Type']) - return None, mime_types + return None, mime_type + return None, [] def _get_mime_type_from_encoding(encoding): @@ -216,15 +194,15 @@ def get_response_class( ) -> Type[Response]: """Guess the most appropriate Response class based on the given arguments.""" - mime_types = list(declared_mime_types or []) + mime_type = next(iter(declared_mime_types or []), None) encoding = None # as in compression (e.g. gzip), not charset if http_headers: - encoding, header_mime_types = ( - _get_encoding_or_mime_types_from_headers(http_headers) + encoding, header_mime_type = ( + _get_encoding_or_mime_type_from_headers(http_headers) ) - if not encoding: - assert header_mime_types is not None - mime_types.extend(header_mime_types) + if encoding is None and mime_type is None: + assert header_mime_type is not None + mime_type = header_mime_type if url is not None: url_parts = urlparse(url) http_origin = url_parts.scheme in ("http", "https") @@ -232,17 +210,15 @@ def get_response_class( encoding, path_mime_type = ( _get_encoding_or_mime_type_from_path(url_parts.path) ) - if not encoding: - mime_types.append(path_mime_type) + if encoding is None and mime_type is None: + mime_type = path_mime_type else: http_origin = True body = _remove_nul_byte_from_text((body or b'')[:BODY_LIMIT]) if encoding: - best_mime_type = _get_mime_type_from_encoding(encoding) - content_types = (best_mime_type,) - elif mime_types: - best_mime_type = _get_best_mime_type(mime_types) - content_types = (best_mime_type,) if best_mime_type else best_mime_type + content_types = (_get_mime_type_from_encoding(encoding),) + elif mime_type: + content_types = (mime_type,) else: content_types = None mime_type = extract_mime( diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index a27e2cc9f..c35ba4025 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -42,6 +42,7 @@ PRE_XTRACTMIME_SCENARIOS = ( ("application/octet-stream", Response), ("text/plain", TextResponse), ("text/html", HtmlResponse), + ("text/html; charset=utf-8", HtmlResponse), ("text/xml", XmlResponse), ) ), @@ -156,13 +157,58 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - ( - { - 'url': 'http://www.example.com/item/', - 'headers': Headers({'Content-Type': ['text/html; charset=utf-8']}), - }, - HtmlResponse, + # Unlike in a web browser, where an attachment Content-Disposition header + # causes the response to be downloaded, and hence MIME sniffing becomes + # irrelevant, in Scrapy those responses are handled the same as any, and + # hence we take the file extension from Content-Disposition into account + # to choose a response class, as a fallback when there is no Content-Type. + *( + ( + { + "url": f"{protocol}://example.com/a", + 'headers': Headers( + { + 'Content-Disposition': [ + f'attachment; filename="a.{file_extension}"', + ] + } + ), + }, + response_class, + ) + for protocol in ("http", "https") + for file_extension, response_class in ( + ("tar.gz", Response), + ("txt", TextResponse), + ("html", HtmlResponse), + ("xml", XmlResponse), + ) ), + *( + ( + { + "url": f"{protocol}://example.com/a", + 'headers': Headers( + { + 'Content-Disposition': [ + f'attachment; filename="a.{file_extension}"', + ], + "Content-Type": {content_type}, + } + ), + }, + response_class, + ) + for protocol in ("http", "https") + for file_extension, content_type, response_class in ( + ("xml", "text/plain", TextResponse), + ("xml", "text/html", HtmlResponse), + ("html", "text/xml", XmlResponse), + ) + ), + + # TODO: Make sure that we have a test that checks that Content-Type + # triumphs body. ( { 'url': 'http://www.example.com/page/', @@ -315,57 +361,6 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - # Compressed content should be of type Response until uncompressed. - # - # When it comes to compression, we trust the encoding from the following - # sources, in the following order: - # - # 1, Content-Encoding HTTP header - # - # 2. File extension of the file name of the Content-Disposition HTTP - # header. - # - # 3. File extension of the file name if the resource comes through a - # file-based protocol (FTP, local file system). - ( - { - 'url': 'file.html', - 'body': b'\n.', - 'headers': Headers( - { - 'Content-Disposition': [ - 'attachment; filename="file.html"' - ], - 'Content-Encoding': ['zip'], - 'Content-Type': ['text/html'], - } - ), - }, - Response, - ), - ( - { - 'url': 'file.html', - 'body': b'\n.', - 'headers': Headers( - { - 'Content-Disposition': [ - 'attachment; filename="file.html.zip"' - ], - 'Content-Type': ['text/html'], - } - ), - }, - Response, - ), - ( - { - 'url': 'file.html.zip', - 'body': b'\n.', - }, - Response, - ), - # We take the file extension of URL paths into account, except for HTTP # responses, because “they are unreliable and easily spoofed”. # @@ -380,6 +375,32 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), + # Unlike in a web browser, where an attachment Content-Disposition header + # causes the response to be downloaded, and hence MIME sniffing becomes + # irrelevant, in Scrapy those responses are handled the same as any, and + # hence we take the file extension from Content-Disposition into account + # to choose a response class, as a fallback when there is no Content-Type. + *( + ( + { + "url": f"{protocol}://example.com/a", + 'headers': Headers( + { + 'Content-Disposition': [ + f'attachment; filename="a.{file_extension}"', + ], + "Content-Type": {content_type}, + } + ), + }, + response_class, + ) + for protocol in ("http", "https") + for file_extension, content_type, response_class in ( + ("xml", "application/octet-stream", Response), + ) + ), + ( { 'body': b'Some plain text', From b2f2788bd41febc62853bf38351844598d89f823 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 16:36:35 +0100 Subject: [PATCH 54/81] Add tests for binary-vs-text body --- scrapy/utils/response.py | 18 +------ tests/test_utils_response.py | 94 +++++++++++++++++++----------------- 2 files changed, 50 insertions(+), 62 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index ea6b33f96..3878e4f45 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -141,22 +141,6 @@ def _get_response_class_from_mime_type(mime_type): return Response -def _remove_nul_byte_from_text(text): - """Return the text with removed null byte (b'\x00') if there are no other - binary bytes in the text, otherwise return the text as-is. - - Based on https://github.com/scrapy/scrapy/issues/2481 - """ - for index in range(len(text)): - if ( - text[index:index + 1] != b'\x00' - and is_binary_data(text[index:index + 1]) - ): - return text - - return text.replace(b'\x00', b'') - - def get_base_url(response: TextResponse) -> str: """Return the base url of the given response, joined with the response url""" warn( @@ -214,7 +198,7 @@ def get_response_class( mime_type = path_mime_type else: http_origin = True - body = _remove_nul_byte_from_text((body or b'')[:BODY_LIMIT]) + body = (body or b'')[:BODY_LIMIT] if encoding: content_types = (_get_mime_type_from_encoding(encoding),) elif mime_type: diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index c35ba4025..32e82172a 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -6,6 +6,7 @@ from pathlib import Path from urllib.parse import urlparse import pytest +from xtractmime import BINARY_BYTES, RESOURCE_HEADER_BUFFER_LENGTH from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.http import HtmlResponse, Response, TextResponse, XmlResponse @@ -44,28 +45,33 @@ PRE_XTRACTMIME_SCENARIOS = ( ("text/html", HtmlResponse), ("text/html; charset=utf-8", HtmlResponse), ("text/xml", XmlResponse), + *( + (mime_type, load_object(class_path)) + for mime_type, class_path in ResponseTypes.CLASSES.items() + ), ) ), - # Even if the body is binary, if the Content-Type says it is text, we - # interpret it as text, as long as the Content-Type is not one of the 4 - # affected by the Apache bug. - # - # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata + # Content-Type triumphs body, except for the Apache bug special case. *( ( { - 'body': b'\x00\x01\xff', + 'body': body, 'headers': Headers({'Content-Type': [content_type]}), }, - TextResponse, + response_class, ) - for content_type in ( - 'text/json', - # text/plain variants *not* affected by the Apache bug - 'text/plain; charset=Iso-8859-1', - 'text/plain; charset=utf-8', - 'text/plain; charset=windows-1252', + for body, content_type, response_class in ( + *( + (b'\x00\x01\xff', content_type, TextResponse) + for content_type in ( + 'text/json', + # text/plain variants *not* affected by the Apache bug + 'text/plain; charset=Iso-8859-1', + 'text/plain; charset=utf-8', + 'text/plain; charset=windows-1252', + ) + ), ) ), @@ -129,20 +135,6 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - # We continue to support the hard-coded MIME-to-Response-class mappings - # from scrapy.responsetypes. - *( - ( - { - 'headers': Headers( - {'Content-Type': [mime_type]} - ), - }, - load_object(class_path), - ) - for mime_type, class_path in ResponseTypes.CLASSES.items() - ), - # We take the file extension of URL paths into account, except for HTTP # responses, because “they are unreliable and easily spoofed”. # @@ -207,26 +199,22 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - # TODO: Make sure that we have a test that checks that Content-Type - # triumphs body. - ( - { - 'url': 'http://www.example.com/page/', - 'headers': Headers( - { - 'Content-Disposition': [ - 'attachment; filename="data.xml.gz"', - ] - } + # A body is considered binary if its header (first 1445 bytes) contains any + # binary data byte. + *( + ({"body": body}, response_class) + for body, response_class in ( + *((byte, Response) for byte in BINARY_BYTES[1:]), + # Binary characters at the end of the header still count. + *( + (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + byte, Response) + for byte in BINARY_BYTES[1:] ), - 'body': b'\x01\x02', - }, - Response, - ), - ( - {'body': b'Some plain\0 text data with\0 tabs and null bytes\0'}, - TextResponse, + # Binary characters right after the header do not count. + (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + BINARY_BYTES[0], TextResponse), + ) ), + ( { 'body': b'\x03\x02\xdf\xdd\x23', @@ -401,6 +389,22 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), + # A body is considered binary if its header (first 1445 bytes) contains any + # binary data byte. + *( + ({"body": body}, response_class) + for body, response_class in ( + (BINARY_BYTES[0], Response), + # Binary characters at the end of the header still count. + (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + BINARY_BYTES[0], Response), + # Binary characters right after the header do not count. + *( + (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + byte, TextResponse) + for byte in BINARY_BYTES[1:] + ), + ) + ), + ( { 'body': b'Some plain text', From 711a6093e125942a7ae278d4634b207f04c25926 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 17:02:56 +0100 Subject: [PATCH 55/81] Finish cleaning up PRE_XTRACTMIME_SCENARIOS --- tests/test_utils_response.py | 114 +++++++++++++---------------------- 1 file changed, 41 insertions(+), 73 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 32e82172a..4d2a3ebbb 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -98,6 +98,9 @@ PRE_XTRACTMIME_SCENARIOS = ( 'text/livescript', 'text/x-ecmascript', 'text/x-javascript', + + # Unofficial + 'application/x-javascript', ) ), @@ -112,6 +115,21 @@ PRE_XTRACTMIME_SCENARIOS = ( for content_type in ( 'application/json', 'text/json', + + # Unofficial + 'application/json-amazonui-streaming', + 'application/x-json', + ) + ), + + # Binary MIME types should trigger a Response. + *( + ( + {'headers': Headers({'Content-Type': [content_type]})}, + Response, + ) + for content_type in ( + 'application/pdf', ) ), @@ -170,7 +188,7 @@ PRE_XTRACTMIME_SCENARIOS = ( ) for protocol in ("http", "https") for file_extension, response_class in ( - ("tar.gz", Response), + ("gz", Response), ("txt", TextResponse), ("html", HtmlResponse), ("xml", XmlResponse), @@ -199,86 +217,39 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - # A body is considered binary if its header (first 1445 bytes) contains any - # binary data byte. + # Binary file extensions should trigger a Response. + *( + ( + { + "url": f"file:///a.{extension}", + }, + Response, + ) + for extension in ( + 'pdf', + ) + ), + + # Without anything else, the body determines the response class. *( ({"body": body}, response_class) for body, response_class in ( + (b'Hello', HtmlResponse), + (b'\n.', HtmlResponse), + + # A body is considered binary if its header (first 1445 bytes) + # contains any binary data byte. *((byte, Response) for byte in BINARY_BYTES[1:]), - # Binary characters at the end of the header still count. *( (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + byte, Response) for byte in BINARY_BYTES[1:] ), - # Binary characters right after the header do not count. (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + BINARY_BYTES[0], TextResponse), ) ), - - ( - { - 'body': b'\x03\x02\xdf\xdd\x23', - 'headers': Headers({'Content-Encoding': 'UTF-8'}), - }, - Response, - ), - ( - { - 'body': b'\x00\x01\xff', - 'url': '://www.example.com/item/', - 'headers': Headers({'Content-Type': ['text/plain']}), - }, - TextResponse, - ), - ({'body': b'Hello'}, HtmlResponse), - ({'body': b'\n.'}, HtmlResponse), - ( - { - 'body': b'\x01\x02', - 'headers': Headers({'Content-Type': ['application/pdf']}), - }, - Response, - ), - ( - {'headers': Headers({'Content-Type': ['application/x-json']})}, - TextResponse, - ), - ( - {'headers': Headers({'Content-Type': ['application/x-javascript']})}, - TextResponse, - ), - ( - { - 'headers': Headers( - {'Content-Type': ['application/json-amazonui-streaming']} - ) - }, - TextResponse, - ), - ( - { - 'headers': Headers( - {'Content-Disposition': ['attachment; filename="data.xml.gz"']} - ), - 'url': 'http://www.example.com/page/', - }, - Response, - ), - ({'headers': Headers({'Content-Type': ['application/pdf']})}, Response), - ({'filename': 'file.pdf'}, Response), - ( - { - 'body': b'\n.', - 'url': 'http://www.example.com', - }, - HtmlResponse, - ), ) # Scenarios that work differently with the previously-used, deprecated @@ -443,9 +414,6 @@ def test_get_response_class_http(kwargs, response_class): kwargs = dict(kwargs) if 'headers' in kwargs: kwargs['http_headers'] = kwargs.pop('headers') - if 'filename' in kwargs: - assert 'url' not in kwargs - kwargs['url'] = kwargs.pop('filename') assert get_response_class(**kwargs) == response_class From 827de3d28815996b2a2881acf4a20d780067943b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 17:22:12 +0100 Subject: [PATCH 56/81] Finish cleaning up POST_XTRACTMIME_SCENARIOS --- tests/test_utils_response.py | 61 ++++++++++++++++++------------------ 1 file changed, 31 insertions(+), 30 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 4d2a3ebbb..eb6cb6b75 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -26,6 +26,15 @@ from scrapy.utils.response import ( __doctests__ = ['scrapy.utils.response'] +NON_BINARY_ASCII_BYTES = ( + byte + for byte in ( + bytes([byte]) for byte in range(128) + ) + if byte not in BINARY_BYTES +) + + # Scenarios that work the same with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args PRE_XTRACTMIME_SCENARIOS = ( @@ -243,6 +252,11 @@ PRE_XTRACTMIME_SCENARIOS = ( # A body is considered binary if its header (first 1445 bytes) # contains any binary data byte. *((byte, Response) for byte in BINARY_BYTES[1:]), + *( + (byte, TextResponse) + for byte in NON_BINARY_ASCII_BYTES + if byte not in (b"\x0c", b"\x1b") + ), *( (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + byte, Response) for byte in BINARY_BYTES[1:] @@ -255,6 +269,15 @@ PRE_XTRACTMIME_SCENARIOS = ( # Scenarios that work differently with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args POST_XTRACTMIME_SCENARIOS = ( + # Content-Type triumphs body, except for the Apache bug special case. + ( + { + 'body': b'a', + 'headers': Headers({'Content-Type': ['application/octet-stream']}), + }, + Response, + ), + # A known Apache bug may cause a server to send files with Content-Type set # to "text/plain", "text/plain; charset=ISO-8859-1", # "text/plain; charset=iso-8859-1", or "text/plain; charset=UTF-8", @@ -285,7 +308,6 @@ POST_XTRACTMIME_SCENARIOS = ( # # https://mimesniff.spec.whatwg.org/#identifying-a-resource-with-an-unknown-mime-type ({}, TextResponse), - ({'url': '/tmp/temp^'}, TextResponse), # Body-based PDF detection # @@ -360,46 +382,25 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - # A body is considered binary if its header (first 1445 bytes) contains any - # binary data byte. + # Without anything else, the body determines the response class. *( ({"body": body}, response_class) for body, response_class in ( + # A body is considered binary if its header (first 1445 bytes) + # contains any binary data byte. (BINARY_BYTES[0], Response), - # Binary characters at the end of the header still count. + *((byte, TextResponse) for byte in (b"\x0c", b"\x1b")), (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + BINARY_BYTES[0], Response), - # Binary characters right after the header do not count. *( (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + byte, TextResponse) for byte in BINARY_BYTES[1:] ), + # HTML and XML detection does not allow for unexpected content + # before document start. + (b'a', TextResponse), + (b'a'}, TextResponse), - ({'body': b'this is not Date: Sun, 8 Jan 2023 17:27:55 +0100 Subject: [PATCH 57/81] Fix typing issues --- scrapy/utils/response.py | 3 +-- 1 file changed, 1 insertion(+), 2 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 3878e4f45..ef3116251 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -110,7 +110,7 @@ def _get_encoding_or_mime_type_from_headers( if encoding: return encoding, None return None, mime_type - return None, [] + return None, None def _get_mime_type_from_encoding(encoding): @@ -185,7 +185,6 @@ def get_response_class( _get_encoding_or_mime_type_from_headers(http_headers) ) if encoding is None and mime_type is None: - assert header_mime_type is not None mime_type = header_mime_type if url is not None: url_parts = urlparse(url) From 24e63f3294494321c12e5648e4e12f8092b65cee Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Sun, 8 Jan 2023 17:28:34 +0100 Subject: [PATCH 58/81] Remove unused imports --- scrapy/utils/response.py | 1 - 1 file changed, 1 deletion(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index ef3116251..36aedf1b1 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -28,7 +28,6 @@ from w3lib import html from xtractmime import ( RESOURCE_HEADER_BUFFER_LENGTH as BODY_LIMIT, extract_mime, - is_binary_data, ) from xtractmime.mimegroups import ( is_html_mime_type, From fcc9e9055399d12804fc5e673d00020a7f9ac21b Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 07:30:40 +0100 Subject: [PATCH 59/81] Wrap XHTML with XmlResponse --- scrapy/utils/response.py | 11 +--- tests/test_utils_response.py | 109 +++++++++++++++++++---------------- 2 files changed, 61 insertions(+), 59 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 36aedf1b1..35bf8baf6 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -66,15 +66,6 @@ def _is_compressed_mime_type(mime_type): return mime_type in _ENCODING_MIME_TYPES -def _is_html_mime_type(mime_type): - if mime_type in { - b'application/xhtml+xml', - b'application/vnd.wap.xhtml+xml', - }: - return True - return is_html_mime_type(mime_type) - - def _is_other_text_mime_type(mime_type): return ( mime_type.startswith(b'text/') @@ -131,7 +122,7 @@ def _get_encoding_or_mime_type_from_path(path): def _get_response_class_from_mime_type(mime_type): if not mime_type: return Response - if _is_html_mime_type(mime_type): + if is_html_mime_type(mime_type): return HtmlResponse if is_xml_mime_type(mime_type): return XmlResponse diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index eb6cb6b75..cbbf201c4 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -57,6 +57,12 @@ PRE_XTRACTMIME_SCENARIOS = ( *( (mime_type, load_object(class_path)) for mime_type, class_path in ResponseTypes.CLASSES.items() + if mime_type not in ( + # “Note that XHTML is best parsed as XML” + # https://lxml.de/parsing.html + "application/xhtml+xml", + "application/vnd.wap.xhtml+xml", + ) ), ) ), @@ -143,23 +149,16 @@ PRE_XTRACTMIME_SCENARIOS = ( ), # Compressed content should be of type Response until uncompressed. - *( - ( - { - 'headers': Headers( - { - 'Content-Encoding': ['zip'], - 'Content-Type': [content_type], - } - ) - }, - Response, - ) - for content_type in ( - 'text/html', - 'text/xml', - 'text/plain', - ) + ( + { + 'headers': Headers( + { + 'Content-Encoding': ['zip'], + 'Content-Type': ['text/html'], + } + ) + }, + Response, ), # We take the file extension of URL paths into account, except for HTTP @@ -168,11 +167,17 @@ PRE_XTRACTMIME_SCENARIOS = ( # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata *( ( - {'url': f'{protocol}://example.com/a.html'}, + {'url': f'{protocol}://example.com/a.{extension}'}, response_class, ) - for protocol, response_class in ( - *((protocol, HtmlResponse) for protocol in ("file", "ftp")), + for protocol in ("file", "ftp") + for extension, response_class in ( + ("gz", Response), + ("html", HtmlResponse), + ("json", TextResponse), + ("pdf", Response), + ("txt", TextResponse), + ("xml", XmlResponse), ) ), @@ -188,20 +193,14 @@ PRE_XTRACTMIME_SCENARIOS = ( 'headers': Headers( { 'Content-Disposition': [ - f'attachment; filename="a.{file_extension}"', + f'attachment; filename="a.xml"', ] } ), }, - response_class, + XmlResponse, ) for protocol in ("http", "https") - for file_extension, response_class in ( - ("gz", Response), - ("txt", TextResponse), - ("html", HtmlResponse), - ("xml", XmlResponse), - ) ), *( ( @@ -210,33 +209,15 @@ PRE_XTRACTMIME_SCENARIOS = ( 'headers': Headers( { 'Content-Disposition': [ - f'attachment; filename="a.{file_extension}"', + f'attachment; filename="a.html"', ], - "Content-Type": {content_type}, + "Content-Type": "text/xml", } ), }, - response_class, + XmlResponse, ) for protocol in ("http", "https") - for file_extension, content_type, response_class in ( - ("xml", "text/plain", TextResponse), - ("xml", "text/html", HtmlResponse), - ("html", "text/xml", XmlResponse), - ) - ), - - # Binary file extensions should trigger a Response. - *( - ( - { - "url": f"file:///a.{extension}", - }, - Response, - ) - for extension in ( - 'pdf', - ) ), # Without anything else, the body determines the response class. @@ -269,6 +250,24 @@ PRE_XTRACTMIME_SCENARIOS = ( # Scenarios that work differently with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args POST_XTRACTMIME_SCENARIOS = ( + # Content-Type determines the type for the HTTP protocol. + *( + ( + { + "url": f"{protocol}://example.com/foo", + "headers": Headers({"Content-Type": content_type}), + }, + response_class, + ) + for protocol in ("http", "https") + for content_type, response_class in ( + # “Note that XHTML is best parsed as XML” + # https://lxml.de/parsing.html + ("application/xhtml+xml", XmlResponse), + ("application/vnd.wap.xhtml+xml", XmlResponse), + ) + ), + # Content-Type triumphs body, except for the Apache bug special case. ( { @@ -346,6 +345,18 @@ POST_XTRACTMIME_SCENARIOS = ( # responses, because “they are unreliable and easily spoofed”. # # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata + *( + ( + {'url': f'{protocol}://example.com/a.{extension}'}, + response_class, + ) + for protocol in ("file", "ftp") + for extension, response_class in ( + # “Note that XHTML is best parsed as XML” + # https://lxml.de/parsing.html + ("xhtml", XmlResponse), + ) + ), *( ( {'url': f'{protocol}://example.com/a.html'}, From eef1f127de24a204149b0b23ad3727cbc68aade6 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 07:43:14 +0100 Subject: [PATCH 60/81] test_response_class_choosing_request: update after recent changes --- tests/test_downloader_handlers.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/test_downloader_handlers.py b/tests/test_downloader_handlers.py index 67e00ddf1..a49d562b3 100644 --- a/tests/test_downloader_handlers.py +++ b/tests/test_downloader_handlers.py @@ -26,7 +26,7 @@ from scrapy.core.downloader.handlers.http10 import HTTP10DownloadHandler from scrapy.core.downloader.handlers.http11 import HTTP11DownloadHandler from scrapy.core.downloader.handlers.s3 import S3DownloadHandler from scrapy.exceptions import NotConfigured -from scrapy.http import Headers, HtmlResponse, Request +from scrapy.http import Headers, HtmlResponse, Request, XmlResponse from scrapy.http.response.text import TextResponse from scrapy.spiders import Spider from scrapy.utils.misc import create_instance @@ -439,12 +439,12 @@ class Http11TestCase(HttpTestCase): """Tests choosing of correct response type in case of Content-Type is empty but body contains text. """ - body = b'Some plain text\ndata with tabs\t and null bytes\0' + xml_body = b' Date: Tue, 10 Jan 2023 11:15:56 +0100 Subject: [PATCH 61/81] Extend tests covering body-based response class choice --- tests/test_utils_response.py | 145 +++++++++++++++++++++++++++++++++-- 1 file changed, 140 insertions(+), 5 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index cbbf201c4..869665406 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -2,6 +2,7 @@ import unittest import warnings +from itertools import chain from pathlib import Path from urllib.parse import urlparse @@ -26,6 +27,29 @@ from scrapy.utils.response import ( __doctests__ = ['scrapy.utils.response'] +# https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata +PRE_XTRACTMIME_HTML_STARTS = ( + b' bytes: + """Make odd bytes lowecase and even bytes uppercase. + + >>> crazy_case(b'foobar') + b'fOoBaR' + """ + return b"".join( + bytes([byte]).lower() if index % 2 == 0 else bytes([byte]).upper() + for index, byte in enumerate(value) + ) + # Scenarios that work the same with the previously-used, deprecated # scrapy.responsetypes.responsetypes.from_args @@ -230,6 +275,58 @@ PRE_XTRACTMIME_SCENARIOS = ( # https://codersblock.com/blog/the-smallest-valid-html5-page/ (b'\n.', HtmlResponse), + # https://mimesniff.spec.whatwg.org/#identifying-a-resource-with-an-unknown-mime-type + *( + (prefix + start + b">", HtmlResponse) + for prefix in ( + b"", + *(byte for byte in WHITESPACE_BYTES if byte != b"\x0c"), + ) + for start in ( + set_case(start) + for set_case in (bytes.lower, bytes.upper, crazy_case) + for start in PRE_XTRACTMIME_HTML_STARTS + ) + ), + *( + (prefix + b"", HtmlResponse) + for start in ( + set_case(start) + for set_case in (bytes.lower, bytes.upper, crazy_case) + for start in POST_XTRACTMIME_HTML_STARTS + ) + ), + *( + (start + b" ", HtmlResponse) + for start in ( + set_case(start) + for set_case in (bytes.lower, bytes.upper, crazy_case) + for start in chain( + PRE_XTRACTMIME_HTML_STARTS, + POST_XTRACTMIME_HTML_STARTS, + ) + ) + ), + *( + (b"\x0c" + start + b">", HtmlResponse) + for start in ( + set_case(start) + for set_case in (bytes.lower, bytes.upper, crazy_case) + for start in PRE_XTRACTMIME_HTML_STARTS + ) + ), + (b"\x0c', TextResponse), From 45b95c63d7e8e0f035897d73b117382c97560c76 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 11:30:06 +0100 Subject: [PATCH 62/81] Address issues reported by pylint --- tests/test_utils_response.py | 4 ++-- tox.ini | 2 +- 2 files changed, 3 insertions(+), 3 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 869665406..6c882b386 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -238,7 +238,7 @@ PRE_XTRACTMIME_SCENARIOS = ( 'headers': Headers( { 'Content-Disposition': [ - f'attachment; filename="a.xml"', + 'attachment; filename="a.xml"', ] } ), @@ -254,7 +254,7 @@ PRE_XTRACTMIME_SCENARIOS = ( 'headers': Headers( { 'Content-Disposition': [ - f'attachment; filename="a.html"', + 'attachment; filename="a.html"', ], "Content-Type": "text/xml", } diff --git a/tox.ini b/tox.ini index e1c52be61..348f298c1 100644 --- a/tox.ini +++ b/tox.ini @@ -4,7 +4,7 @@ # and then run "tox" from this directory. [tox] -envlist = security,flake8,py +envlist = security,flake8,pylint,py minversion = 1.7.0 [testenv] From 8908151608872d32879e11574f75e994005b77fc Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 13:01:25 +0100 Subject: [PATCH 63/81] Test that MIME parameters do not break response class choice --- tests/test_utils_response.py | 168 +++++++++++++++++------------------ 1 file changed, 80 insertions(+), 88 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 6c882b386..0ef775ff4 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -88,16 +88,19 @@ PRE_XTRACTMIME_SCENARIOS = ( ( { "url": f"{protocol}://example.com/foo", - "headers": Headers({"Content-Type": content_type}), + "headers": Headers( + {"Content-Type": content_type + content_type_parameters} + ), }, response_class, ) for protocol in ("http", "https") + # Make sure that MIME parameters do not break response class choice. + for content_type_parameters in ("", "; foo=bar") for content_type, response_class in ( ("application/octet-stream", Response), ("text/plain", TextResponse), ("text/html", HtmlResponse), - ("text/html; charset=utf-8", HtmlResponse), ("text/xml", XmlResponse), *( (mime_type, load_object(class_path)) @@ -109,6 +112,57 @@ PRE_XTRACTMIME_SCENARIOS = ( "application/vnd.wap.xhtml+xml", ) ), + + # JavaScript MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#javascript-mime-type + *( + (mime_type, TextResponse) + for mime_type in ( + 'application/javascript', + 'application/x-javascript', + 'text/ecmascript', + 'text/javascript', + 'text/javascript1.0', + 'text/javascript1.1', + 'text/javascript1.2', + 'text/javascript1.3', + 'text/javascript1.4', + 'text/javascript1.5', + 'text/jscript', + 'text/livescript', + 'text/x-ecmascript', + 'text/x-javascript', + + # Unofficial + 'application/x-javascript', + ) + ), + + # JSON MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#json-mime-type + *( + (mime_type, TextResponse) + for mime_type in ( + 'application/json', + 'text/json', + + # Unofficial + 'application/json-amazonui-streaming', + 'application/x-json', + ) + ), + + # Binary MIME types should trigger a Response. + # + # https://mimesniff.spec.whatwg.org/#json-mime-type + *( + (mime_type, Response) + for mime_type in ( + 'application/pdf', + ) + ), ) ), @@ -135,64 +189,6 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - # JavaScript MIME types should trigger a TextResponse. - # - # https://mimesniff.spec.whatwg.org/#javascript-mime-type - *( - ( - {'headers': Headers({'Content-Type': [content_type]})}, - TextResponse, - ) - for content_type in ( - 'application/javascript', - 'application/x-javascript', - 'text/ecmascript', - 'text/javascript', - 'text/javascript1.0', - 'text/javascript1.1', - 'text/javascript1.2', - 'text/javascript1.3', - 'text/javascript1.4', - 'text/javascript1.5', - 'text/jscript', - 'text/livescript', - 'text/x-ecmascript', - 'text/x-javascript', - - # Unofficial - 'application/x-javascript', - ) - ), - - # JSON MIME types should trigger a TextResponse. - # - # https://mimesniff.spec.whatwg.org/#json-mime-type - *( - ( - {'headers': Headers({'Content-Type': [content_type]})}, - TextResponse, - ) - for content_type in ( - 'application/json', - 'text/json', - - # Unofficial - 'application/json-amazonui-streaming', - 'application/x-json', - ) - ), - - # Binary MIME types should trigger a Response. - *( - ( - {'headers': Headers({'Content-Type': [content_type]})}, - Response, - ) - for content_type in ( - 'application/pdf', - ) - ), - # Compressed content should be of type Response until uncompressed. ( { @@ -357,11 +353,35 @@ POST_XTRACTMIME_SCENARIOS = ( response_class, ) for protocol in ("http", "https") + # Make sure that MIME parameters do not break response class choice. + for content_type_parameters in ("", "; foo=bar") for content_type, response_class in ( # “Note that XHTML is best parsed as XML” # https://lxml.de/parsing.html ("application/xhtml+xml", XmlResponse), ("application/vnd.wap.xhtml+xml", XmlResponse), + + # JavaScript MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#javascript-mime-type + *( + (mime_type, TextResponse) + for mime_type in ( + 'application/ecmascript', + 'application/x-ecmascript', + ) + ), + + # JSON MIME types should trigger a TextResponse. + # + # https://mimesniff.spec.whatwg.org/#json-mime-type + *( + (mime_type, TextResponse) + for mime_type in ( + 'application/foo+json', + 'application/ld+json', + ) + ), ) ), @@ -405,34 +425,6 @@ POST_XTRACTMIME_SCENARIOS = ( # https://mimesniff.spec.whatwg.org/#identifying-a-resource-with-an-unknown-mime-type ({}, TextResponse), - # JavaScript MIME types should trigger a TextResponse. - # - # https://mimesniff.spec.whatwg.org/#javascript-mime-type - *( - ( - {'headers': Headers({'Content-Type': [content_type]})}, - TextResponse, - ) - for content_type in ( - 'application/ecmascript', - 'application/x-ecmascript', - ) - ), - - # JSON MIME types should trigger a TextResponse. - # - # https://mimesniff.spec.whatwg.org/#json-mime-type - *( - ( - {'headers': Headers({'Content-Type': [content_type]})}, - TextResponse, - ) - for content_type in ( - 'application/foo+json', - 'application/ld+json', - ) - ), - # We take the file extension of URL paths into account, except for HTTP # responses, because “they are unreliable and easily spoofed”. # From 2178521a79a49207f7a1be32ddc653164654d9c3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 13:41:36 +0100 Subject: [PATCH 64/81] =?UTF-8?q?Make=20sure=20that=20supplied=20=E2=80=9C?= =?UTF-8?q?unknown=E2=80=9D=20MIME=20types=20from=20Content-Type=20are=20i?= =?UTF-8?q?gnored?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- scrapy/utils/response.py | 10 +++++++- tests/test_utils_response.py | 45 ++++++++++++++++++++++++++++++++++++ 2 files changed, 54 insertions(+), 1 deletion(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 35bf8baf6..62accaf17 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -86,7 +86,15 @@ def _get_encoding_or_mime_type_from_headers( encodings = headers.getlist(b'Content-Encoding') if encodings: return encodings[-1], None - if b'Content-Type' in headers: + if ( + b'Content-Type' in headers + and headers[b'Content-Type'].split(b";")[0].strip().lower() not in ( + b"", + b"unknown/unknown", + b"application/unknown", + b"*/*", + ) + ): return None, headers[b'Content-Type'] if b'Content-Disposition' in headers: path = ( diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 0ef775ff4..171ada62d 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -338,6 +338,51 @@ PRE_XTRACTMIME_SCENARIOS = ( (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + BINARY_BYTES[0], TextResponse), ) ), + + # A Content-Type whose essence is "unknown/unknown", "application/unknown", + # or "*/*" has the same effect as no Content-Type being defined. + # + # https://mimesniff.spec.whatwg.org/#mime-type-sniffing-algorithm + *( + ( + { + 'body': b' Date: Tue, 10 Jan 2023 16:33:54 +0100 Subject: [PATCH 65/81] Test handling of feeds mislabeled as HTML --- tests/test_utils_response.py | 65 +++++++++++++++++++++++++----------- 1 file changed, 46 insertions(+), 19 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 171ada62d..8fd5ac4e3 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -166,7 +166,14 @@ PRE_XTRACTMIME_SCENARIOS = ( ) ), - # Content-Type triumphs body, except for the Apache bug special case. + # Content-Type triumphs body, except for: + # + # - Binary content mislabeled as plain text due to an Apache bug + # https://mimesniff.spec.whatwg.org/#check-for-apache-bug-flag + # https://mimesniff.spec.whatwg.org/#rules-for-text-or-binary + # + # - Feeds mislabeled as HTML + # https://mimesniff.spec.whatwg.org/#rules-for-distinguishing-if-a-resource-is-a-feed-or-html *( ( { @@ -430,7 +437,14 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - # Content-Type triumphs body, except for the Apache bug special case. + # Content-Type triumphs body, except for: + # + # - Binary content mislabeled as plain text due to an Apache bug + # https://mimesniff.spec.whatwg.org/#check-for-apache-bug-flag + # https://mimesniff.spec.whatwg.org/#rules-for-text-or-binary + # + # - Feeds mislabeled as HTML + # https://mimesniff.spec.whatwg.org/#rules-for-distinguishing-if-a-resource-is-a-feed-or-html ( { 'body': b'a', @@ -438,29 +452,42 @@ POST_XTRACTMIME_SCENARIOS = ( }, Response, ), - - # A known Apache bug may cause a server to send files with Content-Type set - # to "text/plain", "text/plain; charset=ISO-8859-1", - # "text/plain; charset=iso-8859-1", or "text/plain; charset=UTF-8", - # regardless of the actual file content. - # - # They should be treated as binary if their content is binary, and as - # text/plain otherwise. - # - # https://mimesniff.spec.whatwg.org/#interpreting-the-resource-metadata *( ( { - 'body': b'\x00\x01\xff', + 'body': body, 'headers': Headers({'Content-Type': [content_type]}), }, - Response, + response_class, ) - for content_type in ( - 'text/plain', - 'text/plain; charset=ISO-8859-1', - 'text/plain; charset=iso-8859-1', - 'text/plain; charset=UTF-8', + for body, content_type, response_class in ( + (b'a', 'application/octet-stream', Response), + *( + (b'\x00\x01\xff', content_type, Response) + for content_type in ( + 'text/plain', + 'text/plain; charset=ISO-8859-1', + 'text/plain; charset=iso-8859-1', + 'text/plain; charset=UTF-8', + ) + ), + *( + (body, "text/html", XmlResponse) + for body in ( + b" Date: Tue, 10 Jan 2023 16:38:42 +0100 Subject: [PATCH 66/81] Address flake8 issues --- tests/test_utils_response.py | 10 ++++------ 1 file changed, 4 insertions(+), 6 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 8fd5ac4e3..feda18a16 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -1,5 +1,3 @@ -# flake8: noqa TODO: remove this line - import unittest import warnings from itertools import chain @@ -339,10 +337,10 @@ PRE_XTRACTMIME_SCENARIOS = ( if byte not in (b"\x0c", b"\x1b") ), *( - (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + byte, Response) + (b"a" * (RESOURCE_HEADER_BUFFER_LENGTH - 1) + byte, Response) for byte in BINARY_BYTES[1:] ), - (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + BINARY_BYTES[0], TextResponse), + (b"a" * RESOURCE_HEADER_BUFFER_LENGTH + BINARY_BYTES[0], TextResponse), ) ), @@ -599,9 +597,9 @@ POST_XTRACTMIME_SCENARIOS = ( # contains any binary data byte. (BINARY_BYTES[0], Response), *((byte, TextResponse) for byte in (b"\x0c", b"\x1b")), - (b"a"*(RESOURCE_HEADER_BUFFER_LENGTH-1) + BINARY_BYTES[0], Response), + (b"a" * (RESOURCE_HEADER_BUFFER_LENGTH - 1) + BINARY_BYTES[0], Response), *( - (b"a"*RESOURCE_HEADER_BUFFER_LENGTH + byte, TextResponse) + (b"a" * RESOURCE_HEADER_BUFFER_LENGTH + byte, TextResponse) for byte in BINARY_BYTES[1:] ), From f538252d3bf38f9758a22aec17e0147ca761dd0c Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 16:40:21 +0100 Subject: [PATCH 67/81] =?UTF-8?q?crazy=5Fcase=20=E2=86=92=20odd=5Fcapitali?= =?UTF-8?q?ze?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/test_utils_response.py | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index feda18a16..ba202ac24 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -66,10 +66,10 @@ WHITESPACE_BYTES = ( ) -def crazy_case(value: bytes) -> bytes: +def odd_capitalize(value: bytes) -> bytes: """Make odd bytes lowecase and even bytes uppercase. - >>> crazy_case(b'foobar') + >>> odd_capitalize(b'foobar') b'fOoBaR' """ return b"".join( @@ -285,7 +285,7 @@ PRE_XTRACTMIME_SCENARIOS = ( ) for start in ( set_case(start) - for set_case in (bytes.lower, bytes.upper, crazy_case) + for set_case in (bytes.lower, bytes.upper, odd_capitalize) for start in PRE_XTRACTMIME_HTML_STARTS ) ), @@ -556,7 +556,7 @@ POST_XTRACTMIME_SCENARIOS = ( (start + b">", HtmlResponse) for start in ( set_case(start) - for set_case in (bytes.lower, bytes.upper, crazy_case) + for set_case in (bytes.lower, bytes.upper, odd_capitalize) for start in POST_XTRACTMIME_HTML_STARTS ) ), @@ -564,7 +564,7 @@ POST_XTRACTMIME_SCENARIOS = ( (start + b" ", HtmlResponse) for start in ( set_case(start) - for set_case in (bytes.lower, bytes.upper, crazy_case) + for set_case in (bytes.lower, bytes.upper, odd_capitalize) for start in chain( PRE_XTRACTMIME_HTML_STARTS, POST_XTRACTMIME_HTML_STARTS, @@ -575,7 +575,7 @@ POST_XTRACTMIME_SCENARIOS = ( (b"\x0c" + start + b">", HtmlResponse) for start in ( set_case(start) - for set_case in (bytes.lower, bytes.upper, crazy_case) + for set_case in (bytes.lower, bytes.upper, odd_capitalize) for start in PRE_XTRACTMIME_HTML_STARTS ) ), From 68fccc9f04bea9ecf0d31fb5ca77559bb9174dfb Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 16:45:33 +0100 Subject: [PATCH 68/81] Test that Content-Disposition takes precedence over body --- tests/test_utils_response.py | 17 +++++++++++++++++ 1 file changed, 17 insertions(+) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index ba202ac24..909e75d33 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -265,6 +265,23 @@ PRE_XTRACTMIME_SCENARIOS = ( ) for protocol in ("http", "https") ), + *( + ( + { + "url": f"{protocol}://example.com/a", + "body": b" Date: Tue, 10 Jan 2023 17:00:33 +0100 Subject: [PATCH 69/81] Revert changes on the deprecated decompression module --- scrapy/downloadermiddlewares/decompression.py | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/scrapy/downloadermiddlewares/decompression.py b/scrapy/downloadermiddlewares/decompression.py index a55023e29..e01e9cc76 100644 --- a/scrapy/downloadermiddlewares/decompression.py +++ b/scrapy/downloadermiddlewares/decompression.py @@ -12,7 +12,7 @@ from tempfile import mktemp from warnings import warn from scrapy.exceptions import ScrapyDeprecationWarning -from scrapy.utils.response import get_response_class +from scrapy.responsetypes import responsetypes warn( @@ -45,7 +45,7 @@ class DecompressionMiddleware: return body = tar_file.extractfile(tar_file.members[0]).read() - respcls = get_response_class(url=tar_file.members[0].name, body=body) + respcls = responsetypes.from_args(filename=tar_file.members[0].name, body=body) return response.replace(body=body, cls=respcls) def _is_zip(self, response): @@ -57,7 +57,7 @@ class DecompressionMiddleware: namelist = zip_file.namelist() body = zip_file.read(namelist[0]) - respcls = get_response_class(url=namelist[0], body=body) + respcls = responsetypes.from_args(filename=namelist[0], body=body) return response.replace(body=body, cls=respcls) def _is_gzip(self, response): @@ -67,7 +67,7 @@ class DecompressionMiddleware: except IOError: return - respcls = get_response_class(body=body) + respcls = responsetypes.from_args(body=body) return response.replace(body=body, cls=respcls) def _is_bzip2(self, response): @@ -76,7 +76,7 @@ class DecompressionMiddleware: except IOError: return - respcls = get_response_class(body=body) + respcls = responsetypes.from_args(body=body) return response.replace(body=body, cls=respcls) def process_response(self, request, response, spider): From ee6f5c56b287f14b175fe88118402ccba98b0571 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 17:55:35 +0100 Subject: [PATCH 70/81] Support multi-encoding Content-Encoding headers --- .../downloadermiddlewares/httpcompression.py | 55 ++++++++++-------- tests/sample_data/compressed/html-br-gzip.bin | Bin 0 -> 4050 bytes ...st_downloadermiddleware_httpcompression.py | 40 ++++++++++--- 3 files changed, 64 insertions(+), 31 deletions(-) create mode 100644 tests/sample_data/compressed/html-br-gzip.bin diff --git a/scrapy/downloadermiddlewares/httpcompression.py b/scrapy/downloadermiddlewares/httpcompression.py index 89132c3bc..c0489dd8e 100644 --- a/scrapy/downloadermiddlewares/httpcompression.py +++ b/scrapy/downloadermiddlewares/httpcompression.py @@ -52,30 +52,39 @@ class HttpCompressionMiddleware: def process_response(self, request, response, spider): - if request.method == 'HEAD': + if ( + request.method == 'HEAD' + or not isinstance(response, Response) + or 'Content-Encoding' not in response.headers + ): return response - if isinstance(response, Response): - content_encoding = response.headers.getlist('Content-Encoding') - if content_encoding: - encoding = content_encoding.pop() - decoded_body = self._decode(response.body, encoding.lower()) - if self.stats: - self.stats.inc_value('httpcompression/response_bytes', len(decoded_body), spider=spider) - self.stats.inc_value('httpcompression/response_count', spider=spider) - respcls = get_response_class( - http_headers=response.headers, - url=response.url, - body=decoded_body, - ) - kwargs = dict(cls=respcls, body=decoded_body) - if issubclass(respcls, TextResponse): - # Force recalculating the encoding based on the new, - # decoded (uncompressed) body. - kwargs['encoding'] = None - response = response.replace(**kwargs) - if not content_encoding: - del response.headers['Content-Encoding'] - + header_list = response.headers.getlist('Content-Encoding') + encodings = [ + item.strip() for item in b",".join(header_list).split(b",") + ] + if not encodings: + return response + while encodings: + encoding = encodings.pop() + decoded_body = self._decode(response.body, encoding.lower()) + if encodings: + response.headers['Content-Encoding'] = b",".join(encodings) + else: + del response.headers['Content-Encoding'] + respcls = get_response_class( + http_headers=response.headers, + url=response.url, + body=decoded_body, + ) + kwargs = dict(cls=respcls, body=decoded_body) + if issubclass(respcls, TextResponse): + # Force recalculating the encoding based on the new, + # decoded (uncompressed) body. + kwargs['encoding'] = None + response = response.replace(**kwargs) + if self.stats: + self.stats.inc_value('httpcompression/response_bytes', len(decoded_body), spider=spider) + self.stats.inc_value('httpcompression/response_count', spider=spider) return response def _decode(self, body, encoding): diff --git a/tests/sample_data/compressed/html-br-gzip.bin b/tests/sample_data/compressed/html-br-gzip.bin new file mode 100644 index 0000000000000000000000000000000000000000..57d935008132948703dfbd2157bfcd2f64a88fb7 GIT binary patch literal 4050 zcmV;@4=wN?iwFP!000000|C1aMDSl!Bry_Ui&!nakqA4OMkN4hnY^*0Sxu-6Onw+O zKen|n&_@R^x11`?UHB&Qdv!Xke+to9ngT^wtJq%|MOrZg%Zv== z#^o6gh#VnQu%QqFyu1{XOa8po$-rXyfeRcL%J;v!v0y3%;;MM5Z1IIOJh@vey$Hrc z`cyq#UGSLm0f#fetq&e={Jv~NOX14rvhUEQQkPa_{t#^bL8eW#S; z;XJAludE^^Xs7vb^G$=2<6| zN|&rWH`{#%GOe{%k{{DoUo_jhzno5^D;vEJ@|0ostBf)Hx5W zNISxB$2Fo->b7*WeazBW|7TMt=b=nnD8*gAu(T_GwV`0LEBZbyV{@bHf-h<@+TWnn z6CrIoPvf>sSP>1Y={-EVepzq(bT5^?cS7ZXrYH@_O^xp8oh@m^ksvPx|B4;3-^8Eq>ywvE%XvJoA?A&MId1)NmDqh9Knhu;w6CG(Y zUJ^Guw$*Q?&t;U+1Tr2!jbm_Q;bOTP7DM2y)Jig^7jcdMs+GjUz2rDC{4sXLm;CSJM)BVO6Y#S*k$V8!;K ziN_Q#9oza+;Kr~+t+iaoS4DXB45jX~{P_9f~H}~<<5#>9( zRB>XoVu^P2T6W@iV$*Td7XvzQY)oBi$&Fi+E^{qnG0afvM7xrVug7w>!Q8d!w})fk zmbsYT9X!p(mc7*AVoZtkzZ|&qpsASb*0}#|iwRDxS&gI{k18%h6XqxHYB1cvf{N+Z zY0bp&Y{B0wws3~n|EIT-DzHIkx2R9nicv=|d?4bovr*1z1MWfKW70P=I37%0%1+#t z;rK&u@A~1#Psy-`C$#lWYi)N>I*G$-i`9Cg!*OBjL`JbZ@fzCVjmNfxn$eB!apKZc zhqgxXsG6~g;K1qXp2RU6Iu#EmwxKn}v*e)R1RG9V9@)gS#CEjCiKQu8IUI`%liNY- zcwD{(ORO_|!G)n`ozIv*(G|t7a9G(6NrW`=g+I zb=U@Pi9M?jmgZdc44vZ^gwg1UnzO*r^nUQh@QK$twp0oI77@vY@Wk=lA!iFU6*o~- zT5Ak$EWLB!9=@eR8+@^NC4e94@AH7^w-9lR+Qil!q<&2Nk713$fvHOO-YtcBCZVYirYp*h}xQ8*>idU1;b90ZWk$4su z)OdXlr*#N*E{?IJd>2`Ic_g1Bcj5TPV-uG#Uf&`Z));&^ap|Fh=zwkXEIalpfm;(@ zNjtVX2J`Ass0nTsm>Ow)n6SmoIah4klJQyOxZbPp|1z1xcHMy)4jr_{xWmTK=&;lw(OfKJjmTM0$vGY$N-XyYtHvpzXqB?N;F3$+&JLx9 z<77?2ld#_LE?m0NK|H*)On)DM{d$WqwXUvoTdJ70C)k&!TejgWsI{lv!<-I|c4BZ! z)^-)y-MC@Ja%(zm@0&flHrr?xn79t7$+4hJTd2e4O|VL%&{@qMHne;RiKBB)>zc&j z1Q#wHcc8XVgQiVp5wjV|#db>=!)go0OEXSA2v_~31lRn~6c#?WC0)q&DnO{^xN2`*{J<)`Tw+HvcY zto0-LKIrI0Cvsa5n+~j|+iY{Q*lpaHx?|m&IYTUEiM>}jShzH+?K^Nfq%~;0&CP;R zH=2@Ryk0FhCy=KH^IjGxWF5E9VM>R#R^#P05=|l2|LrdjF&sKmupVv+i51H4=xU>9 z*((@7UA+YSTei7b(9v4=4jUcD{n!E}(5%A1jj7u@tDS3aW!)XjukiO zlur|1MJRO!`dXfg?JBTu2h@j~1qRk5`5e9uU0uadmn>_Mrgs=Ty6#}0Ujm!6=L#7nYu7cU*n~{pg~m{APi{fw5m5Ivtu0beDLM zusyx#j!rsAogHs0fz~7WKJaqm(wt_WqTI&M!r85I@L+0Jf)nUXcM9ayhnWR7hLo>f zFCB(0qTPzn?41CCz?Q znhy307cNaSieXbD^G^js@8lH1eL~DS*j?DS)O4cf!!MIdJQ)&dKD2_0mC*9b){AJ>WODyQ^=|6l#9}#BaJcqTjITrWPRBchY*zv#VJ2 z1P@p2rvt(Ip6b0k$>FsYC8=!D*{nl0w-J7ask<)X>TD0of)5z**r^_qE$ihbGQ5Xx{iQHF-Nc343r^i?9AZ|-+vUlCC>oKxeIZ@~t6FT?* zT(2UtYi*N#)f&S4*4hiQ~TFgsT$K=(lcc zEDyIcIqj*kxR-4kmPeD#TnN7?-;xA3#Qj_xs0kY zYRfjJmT#vMOZ`&QiM!JZ#CiX(m~+~U@m5jV+hf~?3qH@CT?JvME(^2FS>txkN=)iD zCfC%`cFp5O=OJfYRU_@>#74h+GwmVo88w}7d+ThE`RepQ_DvN%mCS~Fh8Oiip7B&c z+KaKt4-?L827Rl87v=YCOBL6B&orvwGS`v|tPk(_uz$KYn%sT#e`%66 zt7{ux%j0sY*5g{6Nc4&3Wo&9WdhZw$vp+-h;T))0-JXfv&eo}R1O&^eu(r{-+0 z$FNPOYCPsz^1xQ-vO4#A(dYet>YQC&+ie+^<*~f3O67&A%CIUcwo6XByjdOFWf+Fr zGAyrq%^~<_^ndFSD>t3YBm15y=egCk9k0u=U0%bq*OU*hXDMIvx(rUg7_L>?wr$Fx z_MHD`YZH^bOLe}$>uK7rxlPAyd(2hXmT5^D@miL8XD1h4PtknMbbDN;$$lzl8;09u z*oI*kYl+0C?>rd5`P{Huuj{2s)~v4mde(~PV_A;ZQ>iR?JT4{YOs}oZyw$O7Irn;P zm+dy~HRp!gZFyXV+jO~B+csxyoBCdpx3fRIODt8w?W|h_Lz=kd0d`a*_NCUuVLDTZLB39Ot(Ex`nm?Y&q+D6y0*)5*_O-l znuR(KV9eZ7cjjgorpGW=hweRBO-F6tbIB>myKBw`$8cGeV_4o=+1Nkl|MAXLgq|~P zW34mcaonoR^*Szfo?Vq~dtHXx<2B66&VqX$-ubv}PETB}Roa$a)4M9RHtp?na>I1H z9J{JLj=fs8YZhl-rscLh<{GkLm~KtRnU-6Io$SM}Y-`DEe3)nK9rz6*0Tj2WKf4bA E05jrCX#fBK literal 0 HcmV?d00001 diff --git a/tests/test_downloadermiddleware_httpcompression.py b/tests/test_downloadermiddleware_httpcompression.py index e0ac80fa2..ee1f48c81 100644 --- a/tests/test_downloadermiddleware_httpcompression.py +++ b/tests/test_downloadermiddleware_httpcompression.py @@ -22,6 +22,7 @@ FORMAT = { 'rawdeflate': ('html-rawdeflate.bin', 'deflate'), 'zlibdeflate': ('html-zlibdeflate.bin', 'deflate'), 'br': ('html-br.bin', 'br'), + 'br,gzip': ('html-br-gzip.bin', 'br,gzip'), # $ zstd raw.html --content-size -o html-zstd-static-content-size.bin 'zstd-static-content-size': ('html-zstd-static-content-size.bin', 'zstd'), # $ zstd raw.html --no-content-size -o html-zstd-static-no-content-size.bin @@ -133,6 +134,37 @@ class HttpCompressionTest(TestCase): self.assertStatsEqual('httpcompression/response_count', 1) self.assertStatsEqual('httpcompression/response_bytes', 74837) + def test_process_response_br_gzip(self): + try: + import brotli # noqa: F401 + except ImportError: + raise SkipTest("no brotli") + response = self._getresponse('br,gzip') + request = response.request + self.assertEqual(response.headers['Content-Encoding'], b'br,gzip') + newresponse = self.mw.process_response(request, response, self.spider) + assert newresponse is not response + assert newresponse.body.startswith(b" Date: Tue, 10 Jan 2023 18:32:27 +0100 Subject: [PATCH 71/81] Fix _get_encoding_or_mime_type_from_headers handling on comma-separated Content-Encoding values --- scrapy/utils/response.py | 5 ++++- tests/test_utils_response.py | 21 +++++++++++++++++++++ 2 files changed, 25 insertions(+), 1 deletion(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 62accaf17..f1efa2b73 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -83,7 +83,10 @@ def _get_encoding_or_mime_type_from_headers( headers: Headers, ) -> Tuple[Optional[bytes], Optional[bytes]]: if b'Content-Encoding' in headers: - encodings = headers.getlist(b'Content-Encoding') + encodings = [ + item.strip() for item in + b",".join(headers.getlist(b'Content-Encoding')).split(b",") + ] if encodings: return encodings[-1], None if ( diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 909e75d33..0f8e99999 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -19,6 +19,7 @@ from scrapy.utils.response import ( open_in_browser, response_httprepr, response_status_message, + _get_encoding_or_mime_type_from_headers, ) @@ -643,6 +644,26 @@ def test_get_response_class_http(kwargs, response_class): assert get_response_class(**kwargs) == response_class +@pytest.mark.parametrize( + 'headers,expected', + ( + *( + ( + Headers({'Content-Encoding': content_encoding_header}), + (encoding, None), + ) + for content_encoding_header, encoding in ( + (['gzip'], b'gzip'), + (['gzip', 'compress'], b'compress'), + (['deflate, br'], b'br'), + ) + ), + ), +) +def test_get_encoding_or_mime_type_from_headers(headers, expected): + assert _get_encoding_or_mime_type_from_headers(headers) == expected + + class ResponseUtilsTest(unittest.TestCase): dummy_response = TextResponse(url='http://example.org/', body=b'dummy_response') From deab93b243248eec77f460ddd1bc0eac8cae0133 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 19:04:52 +0100 Subject: [PATCH 72/81] Remove duplicate test --- tests/test_utils_response.py | 7 ------- 1 file changed, 7 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 0f8e99999..c4a66184d 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -461,13 +461,6 @@ POST_XTRACTMIME_SCENARIOS = ( # # - Feeds mislabeled as HTML # https://mimesniff.spec.whatwg.org/#rules-for-distinguishing-if-a-resource-is-a-feed-or-html - ( - { - 'body': b'a', - 'headers': Headers({'Content-Type': ['application/octet-stream']}), - }, - Response, - ), *( ( { From ff37b274a66f1975ea99d635918da69e7aabb81e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 20:09:16 +0100 Subject: [PATCH 73/81] Remove unused variable --- scrapy/utils/response.py | 1 - 1 file changed, 1 deletion(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index f1efa2b73..4e0d31736 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -48,7 +48,6 @@ from scrapy.http import ( from scrapy.utils.decorators import deprecated from scrapy.utils.python import to_bytes, to_unicode -_baseurl_cache: "WeakKeyDictionary[Response, str]" = WeakKeyDictionary() _ENCODING_MIME_TYPE_MAP = { b'br': b'application/brotli', b'compress': b'application/x-compress', From 86baef02a04dd5566e5ad8568011223cc29723f3 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 20:51:43 +0100 Subject: [PATCH 74/81] Test Content-Encoding against Content-Disposition, URL and body --- tests/test_utils_response.py | 37 ++++++++++++++++++++++++++++++++++++ 1 file changed, 37 insertions(+) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index c4a66184d..53366d428 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -500,6 +500,43 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), + # Compressed content should be of type Response until uncompressed. + ( + { + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="a.html"', + ], + 'Content-Encoding': ['zip'], + } + ) + }, + Response, + ), + ( + { + "url": "file:///a.html", + 'headers': Headers( + { + 'Content-Encoding': ['zip'], + } + ) + }, + Response, + ), + ( + { + "body": b"", + 'headers': Headers( + { + 'Content-Encoding': ['zip'], + } + ) + }, + Response, + ), + # If the body is empty, it contains no binary data bytes, hence body-based # MIME type detection must interpret the result as text. # From 7bb5289cedc2a7410c58e2c4fc8dfda692faaa9a Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 21:11:48 +0100 Subject: [PATCH 75/81] Cover additional behavior changes in tests --- tests/test_utils_response.py | 58 +++++++++++++++++++++++++++++------- 1 file changed, 47 insertions(+), 11 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 53366d428..0caa7b051 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -450,6 +450,16 @@ POST_XTRACTMIME_SCENARIOS = ( 'application/ld+json', ) ), + + # XML MIME types should trigger an XmlResponse. + # + # https://mimesniff.spec.whatwg.org/#xml-mime-type + *( + (mime_type, XmlResponse) + for mime_type in ( + 'application/foo+xml', + ) + ), ) ), @@ -500,6 +510,21 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), + # Content-Type also triumphs Content-Disposition. + ( + { + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="a.html"', + ], + 'Content-Type': ['application/octet-stream'], + } + ) + }, + Response, + ), + # Compressed content should be of type Response until uncompressed. ( { @@ -514,17 +539,6 @@ POST_XTRACTMIME_SCENARIOS = ( }, Response, ), - ( - { - "url": "file:///a.html", - 'headers': Headers( - { - 'Content-Encoding': ['zip'], - } - ) - }, - Response, - ), ( { "body": b"", @@ -595,6 +609,28 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), + # File extension triumphs body. + ( + { + "body": b"", + 'headers': Headers( + { + 'Content-Disposition': [ + 'attachment; filename="a.gz"', + ], + } + ) + }, + Response, + ), + ( + { + "body": b"", + 'url': 'file:///a.gz', + }, + Response, + ), + # Without anything else, the body determines the response class. *( ({"body": body}, response_class) From 9ee2ebea54c7c5e6a29e373eb083f8066d185521 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 21:27:58 +0100 Subject: [PATCH 76/81] Document TextResponse.base_url --- docs/topics/request-response.rst | 2 ++ scrapy/http/response/text.py | 8 +++++++- 2 files changed, 9 insertions(+), 1 deletion(-) diff --git a/docs/topics/request-response.rst b/docs/topics/request-response.rst index a0d9fc03e..7af180bdf 100644 --- a/docs/topics/request-response.rst +++ b/docs/topics/request-response.rst @@ -1203,6 +1203,8 @@ TextResponse objects .. autoattribute:: TextResponse.attributes + .. autoattribute:: TextResponse.base_url + :class:`TextResponse` objects support the following methods in addition to the standard :class:`Response` ones: diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 8b12578d3..6e3f8a926 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -95,7 +95,13 @@ class TextResponse(Response): @property def base_url(self) -> str: - """Base URL""" + """Base URL for any relative URL in the response. + + It defaults to the response :attr:`~scrapy.http.Response.url`, but HTML + responses may include a `base element + `_ with + a different base URL. + """ if self._cached_base_url is None: self._cached_base_url = get_base_url( self.text[:4096], From f8cdd092d5ac8a3a0e987713cc05c6aa86087369 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 21:32:00 +0100 Subject: [PATCH 77/81] Deprecate binary_is_text --- scrapy/utils/python.py | 10 ++++++++++ 1 file changed, 10 insertions(+) diff --git a/scrapy/utils/python.py b/scrapy/utils/python.py index 9df1c91de..983e3e322 100644 --- a/scrapy/utils/python.py +++ b/scrapy/utils/python.py @@ -9,7 +9,9 @@ import weakref from functools import partial, wraps from itertools import chain from typing import AsyncGenerator, AsyncIterable, Iterable, Union +from warnings import warn +from scrapy.exceptions import ScrapyDeprecationWarning from scrapy.utils.asyncgen import as_async_generator @@ -165,6 +167,14 @@ def binary_is_text(data): """ Returns ``True`` if the given ``data`` argument (a ``bytes`` object) does not contain unprintable control characters. """ + warn( + ( + 'scrapy.utils.python.binary_is_text is deprecated, use ' + 'xtractmime.is_binary_data instead.' + ), + ScrapyDeprecationWarning, + stacklevel=2, + ) if not isinstance(data, bytes): raise TypeError(f"data must be bytes, got '{type(data).__name__}'") return all(c not in _BINARYCHARS for c in data) From 1ade9d2ed7dbbd268a75fd6b4228b868a48ee5d9 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Tue, 10 Jan 2023 21:41:49 +0100 Subject: [PATCH 78/81] Minor test changes --- tests/test_utils_response.py | 8 ++------ 1 file changed, 2 insertions(+), 6 deletions(-) diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 0caa7b051..c056d3682 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -583,11 +583,7 @@ POST_XTRACTMIME_SCENARIOS = ( ) ), - # Unlike in a web browser, where an attachment Content-Disposition header - # causes the response to be downloaded, and hence MIME sniffing becomes - # irrelevant, in Scrapy those responses are handled the same as any, and - # hence we take the file extension from Content-Disposition into account - # to choose a response class, as a fallback when there is no Content-Type. + # Content-Type triumphes Content-Disposition. *( ( { @@ -597,7 +593,7 @@ POST_XTRACTMIME_SCENARIOS = ( 'Content-Disposition': [ f'attachment; filename="a.{file_extension}"', ], - "Content-Type": {content_type}, + "Content-Type": [content_type], } ), }, From 13bc1499ac2191d39bf9643e6c3b3770a2207240 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Adri=C3=A1n=20Chaves?= Date: Fri, 9 Feb 2024 08:20:49 +0100 Subject: [PATCH 79/81] Ignore Content-Type when it triggers Response --- scrapy/utils/response.py | 11 +++ tests/test_responsetypes.py | 4 +- tests/test_utils_response.py | 135 ++++++++++++++++++++++++----------- 3 files changed, 108 insertions(+), 42 deletions(-) diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 3459b47bc..2ab167478 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -203,6 +203,17 @@ def get_response_class( content_types=content_types, http_origin=http_origin, ) + cls = _get_response_class_from_mime_type(mime_type) + if cls is not Response or not content_types or encoding or not http_origin: + return cls + # In scenarios where there was a declared Content-Type, no + # Content-Encoding, HTTP/HTTPS was used, and xtractmime determined the + # output to be binary, repeat MIME extraction ignoring the declared + # Content-Type, so that the body is taken into account. + mime_type = extract_mime( + body, + http_origin=http_origin, + ) return _get_response_class_from_mime_type(mime_type) diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 7d769cfda..a19a2133f 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -45,7 +45,9 @@ def _unmark(item): ), ) def test_from_args(kwargs, response_class): - assert responsetypes.from_args(**kwargs) == response_class + assert ( + responsetypes.from_args(**kwargs) == response_class + ), f"{responsetypes.from_args(**kwargs)=} != {response_class=}" class ResponseTypesTest(unittest.TestCase): diff --git a/tests/test_utils_response.py b/tests/test_utils_response.py index 838169e4f..f226d6c88 100644 --- a/tests/test_utils_response.py +++ b/tests/test_utils_response.py @@ -89,7 +89,6 @@ PRE_XTRACTMIME_SCENARIOS = ( # Make sure that MIME parameters do not break response class choice. for content_type_parameters in ("", "; foo=bar") for content_type, response_class in ( - ("application/octet-stream", Response), ("text/plain", TextResponse), ("text/html", HtmlResponse), ("text/xml", XmlResponse), @@ -140,10 +139,6 @@ PRE_XTRACTMIME_SCENARIOS = ( "application/x-json", ) ), - # Binary MIME types should trigger a Response. - # - # https://mimesniff.spec.whatwg.org/#json-mime-type - *((mime_type, Response) for mime_type in ("application/pdf",)), ) ), # Content-Type triumphs body, except for: @@ -174,6 +169,28 @@ PRE_XTRACTMIME_SCENARIOS = ( ), ) ), + # Content-Type triumphs Content-Disposition. + *( + ( + { + "url": f"{protocol}://example.com/a", + "headers": Headers( + { + "Content-Disposition": [ + f'attachment; filename="a.{file_extension}"', + ], + "Content-Type": [content_type], + } + ), + }, + response_class, + ) + for protocol in ("http", "https") + for file_extension, content_type, response_class in ( + ("html", "application/json", JsonResponse), + ("xml", "application/json", JsonResponse), + ) + ), # Compressed content should be of type Response until uncompressed. ( { @@ -378,6 +395,36 @@ PRE_XTRACTMIME_SCENARIOS = ( "*/*", ) ), + # Content triumphs Content-Type when using HTTP or HTTPS and the + # Content-Type is unknown or binary while the content is plain text. This + # is a conscious divergence from the MIME Sniffing Standard for a better + # web scraping experience. + *( + ( + { + "url": f"{protocol}://example.com/foo", + "headers": Headers({"Content-Type": content_type}), + "body": body, + }, + TextResponse, + ) + for protocol in ("http", "https") + for body in ( + b"", + b"a", + b"var a = 'b';", + b'{"a": "b"}', + b'.a {b: "c"}', + ) + for content_type in ( + "application/octet-stream", + "application/pdf", + "application/custom", + "application/bad-custom-json", # Should end in +json + "application/bad-custom-text", # Should start with text/ + "application/bad-custom-xml", # Should end in +xml + ) + ), ) # Scenarios that work differently with the previously-used, deprecated @@ -444,7 +491,6 @@ POST_XTRACTMIME_SCENARIOS = ( response_class, ) for body, content_type, response_class in ( - (b"a", "application/octet-stream", Response), *( (b"\x00\x01\xff", content_type, Response) for content_type in ( @@ -474,20 +520,6 @@ POST_XTRACTMIME_SCENARIOS = ( ), ) ), - # Content-Type also triumphs Content-Disposition. - ( - { - "headers": Headers( - { - "Content-Disposition": [ - 'attachment; filename="a.html"', - ], - "Content-Type": ["application/octet-stream"], - } - ) - }, - Response, - ), # Compressed content should be of type Response until uncompressed. ( { @@ -543,27 +575,6 @@ POST_XTRACTMIME_SCENARIOS = ( *((protocol, TextResponse) for protocol in ("http", "https")), ) ), - # Content-Type triumphes Content-Disposition. - *( - ( - { - "url": f"{protocol}://example.com/a", - "headers": Headers( - { - "Content-Disposition": [ - f'attachment; filename="a.{file_extension}"', - ], - "Content-Type": [content_type], - } - ), - }, - response_class, - ) - for protocol in ("http", "https") - for file_extension, content_type, response_class in ( - ("xml", "application/octet-stream", Response), - ) - ), # File extension triumphs body. ( { @@ -645,6 +656,48 @@ POST_XTRACTMIME_SCENARIOS = ( (b"a Date: Wed, 10 Jun 2026 11:13:24 +0200 Subject: [PATCH 80/81] Ignore Content-Type when it triggers Response `identity` in `Content-Encoding` means no transformation (RFC 7231), so it should not be treated as a compression encoding when sniffing the MIME type. Previously it was mapped to `application/identity`, causing `get_response_class` to return `Response` for HTML bodies, which then broke `response.replace(cls=Response)` on `TextResponse` objects (encoding is not a valid `Response.__init__` kwarg). Co-Authored-By: Claude Sonnet 4.6 --- scrapy/http/response/text.py | 1 + scrapy/utils/response.py | 1 + tests/test_downloader_handlers_http_base.py | 2 +- 3 files changed, 3 insertions(+), 1 deletion(-) diff --git a/scrapy/http/response/text.py b/scrapy/http/response/text.py index 695de5482..b77f13a44 100644 --- a/scrapy/http/response/text.py +++ b/scrapy/http/response/text.py @@ -43,6 +43,7 @@ class TextResponse(Response): attributes: tuple[str, ...] = (*Response.attributes, "encoding") __slots__ = ( + "_cached_base_url", "_cached_benc", "_cached_decoded_json", "_cached_selector", diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 48ef9b1ff..470f6431f 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -78,6 +78,7 @@ def _get_encoding_or_mime_type_from_headers( encodings = [ item.strip() for item in b",".join(headers.getlist(b"Content-Encoding")).split(b",") + if item.strip().lower() != b"identity" ] if encodings: return encodings[-1], None diff --git a/tests/test_downloader_handlers_http_base.py b/tests/test_downloader_handlers_http_base.py index 0e1ff07c9..233264c03 100644 --- a/tests/test_downloader_handlers_http_base.py +++ b/tests/test_downloader_handlers_http_base.py @@ -549,7 +549,7 @@ class TestHttpBase(ABC): """Tests choosing of correct response type in case of Content-Type is empty but body contains text. """ - body = b"Some plain text\ndata with tabs\t and null bytes\0" + body = b"Some plain text\ndata with tabs\t" request = Request( mockserver.url("/nocontenttype", is_secure=self.is_secure), body=body ) From 82e280dde19e70d0bce2809813deef2bad33e506 Mon Sep 17 00:00:00 2001 From: Adrian Chaves Date: Wed, 10 Jun 2026 12:05:07 +0200 Subject: [PATCH 81/81] Fix mypy and pylint issues introduced by the xtractmime merge - Add xtractmime to the mypy ignore_missing_imports overrides (no py.typed marker in that library) - Narrow Content-Disposition header via a local variable so mypy can see it is non-None inside the if-block - Remove a redundant local re-import of HtmlResponse/TextResponse inside open_in_browser (they are already imported at module level, fixing pylint W0404 reimported) - Add Any annotations to _unmark() in test_responsetypes.py so mypy does not raise no-untyped-call when it is called from a typed context Co-Authored-By: Claude Sonnet 4.6 --- pyproject.toml | 2 ++ scrapy/utils/response.py | 12 +++--------- tests/test_responsetypes.py | 6 +++++- 3 files changed, 10 insertions(+), 10 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 50322b17a..8cd7310ea 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -133,6 +133,8 @@ module = [ "pytest_twisted", "robotexclusionrulesparser", "testfixtures", + "xtractmime", + "xtractmime.*", "zope.interface.*", ] ignore_missing_imports = true diff --git a/scrapy/utils/response.py b/scrapy/utils/response.py index 470f6431f..a27ad62dc 100644 --- a/scrapy/utils/response.py +++ b/scrapy/utils/response.py @@ -94,13 +94,10 @@ def _get_encoding_or_mime_type_from_headers( ) ): return None, headers[b"Content-Type"] - if headers.get(b"Content-Disposition"): + content_disposition = headers.get(b"Content-Disposition") + if content_disposition: path = ( - headers[b"Content-Disposition"] - .split(b";")[-1] - .split(b"=")[-1] - .strip(b"\"'") - .decode() + content_disposition.split(b";")[-1].split(b"=")[-1].strip(b"\"'").decode() ) encoding, mime_type = _get_encoding_or_mime_type_from_path(path) if encoding: @@ -262,9 +259,6 @@ def open_in_browser( if "item name" not in response.body: open_in_browser(response) """ - # circular imports - from scrapy.http import HtmlResponse, TextResponse # noqa: PLC0415 - # XXX: this implementation is a bit dirty and could be improved body = response.body if isinstance(response, HtmlResponse): diff --git a/tests/test_responsetypes.py b/tests/test_responsetypes.py index 3b71567b8..73b9aaf2b 100644 --- a/tests/test_responsetypes.py +++ b/tests/test_responsetypes.py @@ -1,3 +1,7 @@ +from __future__ import annotations + +from typing import Any + import pytest from scrapy.http import ( @@ -13,7 +17,7 @@ from scrapy.responsetypes import responsetypes from .test_utils_response import POST_XTRACTMIME_SCENARIOS, PRE_XTRACTMIME_SCENARIOS -def _unmark(item): +def _unmark(item: Any) -> Any: return pytest.param(*item.values)