mirror of https://github.com/scrapy/scrapy.git
Provide complete API documentation coverage of scrapy.linkextractors
This commit is contained in:
parent
2c14692e60
commit
7f4f98fd38
|
|
@ -265,6 +265,10 @@ coverage_ignore_pyobjects = [
|
|||
|
||||
# Never documented before, and deprecated now.
|
||||
r'^scrapy\.item\.DictItem$',
|
||||
r'^scrapy\.linkextractors\.FilteringLinkExtractor$',
|
||||
|
||||
# Implementation detail of LxmlLinkExtractor
|
||||
r'^scrapy\.linkextractors\.lxmlhtml\.LxmlParserLinkExtractor',
|
||||
]
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -4,46 +4,33 @@
|
|||
Link Extractors
|
||||
===============
|
||||
|
||||
Link extractors are objects whose only purpose is to extract links from web
|
||||
pages (:class:`scrapy.http.Response` objects) which will be eventually
|
||||
followed.
|
||||
A link extractor is an object that extracts links from responses.
|
||||
|
||||
There is ``scrapy.linkextractors.LinkExtractor`` available
|
||||
in Scrapy, but you can create your own custom Link Extractors to suit your
|
||||
needs by implementing a simple interface.
|
||||
|
||||
The only public method that every link extractor has is ``extract_links``,
|
||||
which receives a :class:`~scrapy.http.Response` object and returns a list
|
||||
of :class:`scrapy.link.Link` objects. Link extractors are meant to be
|
||||
instantiated once and their ``extract_links`` method called several times
|
||||
with different responses to extract links to follow.
|
||||
|
||||
Link extractors are used in the :class:`~scrapy.spiders.CrawlSpider`
|
||||
class (available in Scrapy), through a set of rules, but you can also use it in
|
||||
your spiders, even if you don't subclass from
|
||||
:class:`~scrapy.spiders.CrawlSpider`, as its purpose is very simple: to
|
||||
extract links.
|
||||
The constructor of :class:`~scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`
|
||||
takes settings that determine which links may be extracted.
|
||||
:class:`LxmlLinkExtractor.extract_links
|
||||
<scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor.extract_links>` returns a
|
||||
list of matching :class:`scrapy.link.Link` objects from a
|
||||
:class:`~scrapy.http.Response` object.
|
||||
|
||||
Link extractors are used in :class:`~scrapy.spiders.CrawlSpider` spiders
|
||||
through a set of :class:`~scrapy.spiders.Rule` objects. You can also use link
|
||||
extractors in regular spiders.
|
||||
|
||||
.. _topics-link-extractors-ref:
|
||||
|
||||
Built-in link extractors reference
|
||||
==================================
|
||||
Link extractor reference
|
||||
========================
|
||||
|
||||
.. module:: scrapy.linkextractors
|
||||
:synopsis: Link extractors classes
|
||||
|
||||
Link extractors classes bundled with Scrapy are provided in the
|
||||
:mod:`scrapy.linkextractors` module.
|
||||
|
||||
The default link extractor is ``LinkExtractor``, which is the same as
|
||||
:class:`~.LxmlLinkExtractor`::
|
||||
The link extractor class is
|
||||
:class:`scrapy.linkextractors.lxmlhtml.LxmlLinkExtractor`. For convenience it
|
||||
can also be imported as ``scrapy.linkextractors.LinkExtractor``::
|
||||
|
||||
from scrapy.linkextractors import LinkExtractor
|
||||
|
||||
There used to be other link extractor classes in previous Scrapy versions,
|
||||
but they are deprecated now.
|
||||
|
||||
LxmlLinkExtractor
|
||||
-----------------
|
||||
|
||||
|
|
@ -152,4 +139,6 @@ LxmlLinkExtractor
|
|||
from elements or attributes which allow leading/trailing whitespaces).
|
||||
:type strip: boolean
|
||||
|
||||
.. automethod:: extract_links
|
||||
|
||||
.. _scrapy.linkextractors: https://github.com/scrapy/scrapy/blob/master/scrapy/linkextractors/__init__.py
|
||||
|
|
|
|||
|
|
@ -6,11 +6,13 @@ This package contains a collection of Link Extractors.
|
|||
For more info see docs/topics/link-extractors.rst
|
||||
"""
|
||||
import re
|
||||
from warnings import warn
|
||||
|
||||
from six.moves.urllib.parse import urlparse
|
||||
from parsel.csstranslator import HTMLTranslator
|
||||
from w3lib.url import canonicalize_url
|
||||
|
||||
from scrapy.utils.deprecate import ScrapyDeprecationWarning
|
||||
from scrapy.utils.misc import arg_to_iter
|
||||
from scrapy.utils.url import (
|
||||
url_is_from_any_domain, url_has_any_extension,
|
||||
|
|
@ -49,6 +51,15 @@ class FilteringLinkExtractor(object):
|
|||
|
||||
_csstranslator = HTMLTranslator()
|
||||
|
||||
def __new__(cls, *args, **kwargs):
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
|
||||
if (issubclass(cls, FilteringLinkExtractor) and
|
||||
not issubclass(cls, LxmlLinkExtractor)):
|
||||
warn('scrapy.linkextractors.FilteringLinkExtractor is deprecated, '
|
||||
'please use scrapy.linkextractors.LinkExtractor instead',
|
||||
ScrapyDeprecationWarning, stacklevel=2)
|
||||
return super(FilteringLinkExtractor, cls).__new__(cls)
|
||||
|
||||
def __init__(self, link_extractor, allow, deny, allow_domains, deny_domains,
|
||||
restrict_xpaths, canonicalize, deny_extensions, restrict_css, restrict_text):
|
||||
|
||||
|
|
|
|||
|
|
@ -117,6 +117,14 @@ class LxmlLinkExtractor(FilteringLinkExtractor):
|
|||
restrict_text=restrict_text)
|
||||
|
||||
def extract_links(self, response):
|
||||
"""Returns a list of :class:`~scrapy.link.Link` objects from the
|
||||
specified :class:`response <scrapy.http.Response>`.
|
||||
|
||||
Only links that match the settings passed to the link extractor
|
||||
constructor are returned.
|
||||
|
||||
Duplicate links are omitted.
|
||||
"""
|
||||
base_url = get_base_url(response)
|
||||
if self.restrict_xpaths:
|
||||
docs = [subdoc
|
||||
|
|
|
|||
|
|
@ -1,10 +1,13 @@
|
|||
import re
|
||||
import unittest
|
||||
from warnings import catch_warnings
|
||||
|
||||
import pytest
|
||||
|
||||
from scrapy.exceptions import ScrapyDeprecationWarning
|
||||
from scrapy.http import HtmlResponse, XmlResponse
|
||||
from scrapy.link import Link
|
||||
from scrapy.linkextractors import FilteringLinkExtractor
|
||||
from scrapy.linkextractors.lxmlhtml import LxmlLinkExtractor
|
||||
from tests import get_testdata
|
||||
|
||||
|
|
@ -506,3 +509,30 @@ class LxmlLinkExtractorTestCase(Base.LinkExtractorTestCase):
|
|||
@pytest.mark.xfail
|
||||
def test_restrict_xpaths_with_html_entities(self):
|
||||
super(LxmlLinkExtractorTestCase, self).test_restrict_xpaths_with_html_entities()
|
||||
|
||||
def test_filteringlinkextractor_deprecation_warning(self):
|
||||
"""Make sure the FilteringLinkExtractor deprecation warning is not
|
||||
issued for LxmlLinkExtractor"""
|
||||
with catch_warnings(record=True) as warnings:
|
||||
extractor = LxmlLinkExtractor()
|
||||
self.assertEqual(len(warnings), 0)
|
||||
class SubclassedItem(LxmlLinkExtractor):
|
||||
pass
|
||||
subclassed_extractor = SubclassedItem()
|
||||
self.assertEqual(len(warnings), 0)
|
||||
|
||||
|
||||
class FilteringLinkExtractorTest(unittest.TestCase):
|
||||
|
||||
def test_deprecation_warning(self):
|
||||
args = [None]*10
|
||||
with catch_warnings(record=True) as warnings:
|
||||
extractor = FilteringLinkExtractor(*args)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
with catch_warnings(record=True) as warnings:
|
||||
class SubclassedFilteringLinkExtractor(FilteringLinkExtractor):
|
||||
pass
|
||||
subclassed_extractor = SubclassedFilteringLinkExtractor(*args)
|
||||
self.assertEqual(len(warnings), 1)
|
||||
self.assertEqual(warnings[0].category, ScrapyDeprecationWarning)
|
||||
|
|
|
|||
Loading…
Reference in New Issue