1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180
|
"""
Some spiders used for testing and benchmarking
"""
import time
from six.moves.urllib.parse import urlencode
from scrapy.spiders import Spider
from scrapy.http import Request
from scrapy.item import Item
from scrapy.linkextractors import LinkExtractor
class MetaSpider(Spider):
name = 'meta'
def __init__(self, *args, **kwargs):
super(MetaSpider, self).__init__(*args, **kwargs)
self.meta = {}
def closed(self, reason):
self.meta['close_reason'] = reason
class FollowAllSpider(MetaSpider):
name = 'follow'
link_extractor = LinkExtractor()
def __init__(self, total=10, show=20, order="rand", maxlatency=0.0, *args, **kwargs):
super(FollowAllSpider, self).__init__(*args, **kwargs)
self.urls_visited = []
self.times = []
qargs = {'total': total, 'show': show, 'order': order, 'maxlatency': maxlatency}
url = "http://localhost:8998/follow?%s" % urlencode(qargs, doseq=1)
self.start_urls = [url]
def parse(self, response):
self.urls_visited.append(response.url)
self.times.append(time.time())
for link in self.link_extractor.extract_links(response):
yield Request(link.url, callback=self.parse)
class DelaySpider(MetaSpider):
name = 'delay'
def __init__(self, n=1, b=0, *args, **kwargs):
super(DelaySpider, self).__init__(*args, **kwargs)
self.n = n
self.b = b
self.t1 = self.t2 = self.t2_err = 0
def start_requests(self):
self.t1 = time.time()
url = "http://localhost:8998/delay?n=%s&b=%s" % (self.n, self.b)
yield Request(url, callback=self.parse, errback=self.errback)
def parse(self, response):
self.t2 = time.time()
def errback(self, failure):
self.t2_err = time.time()
class SimpleSpider(MetaSpider):
name = 'simple'
def __init__(self, url="http://localhost:8998", *args, **kwargs):
super(SimpleSpider, self).__init__(*args, **kwargs)
self.start_urls = [url]
def parse(self, response):
self.logger.info("Got response %d" % response.status)
class ItemSpider(FollowAllSpider):
name = 'item'
def parse(self, response):
for request in super(ItemSpider, self).parse(response):
yield request
yield Item()
yield {}
class DefaultError(Exception):
pass
class ErrorSpider(FollowAllSpider):
name = 'error'
exception_cls = DefaultError
def raise_exception(self):
raise self.exception_cls('Expected exception')
def parse(self, response):
for request in super(ErrorSpider, self).parse(response):
yield request
self.raise_exception()
class BrokenStartRequestsSpider(FollowAllSpider):
fail_before_yield = False
fail_yielding = False
def __init__(self, *a, **kw):
super(BrokenStartRequestsSpider, self).__init__(*a, **kw)
self.seedsseen = []
def start_requests(self):
if self.fail_before_yield:
1 / 0
for s in range(100):
qargs = {'total': 10, 'seed': s}
url = "http://localhost:8998/follow?%s" % urlencode(qargs, doseq=1)
yield Request(url, meta={'seed': s})
if self.fail_yielding:
2 / 0
assert self.seedsseen, \
'All start requests consumed before any download happened'
def parse(self, response):
self.seedsseen.append(response.meta.get('seed'))
for req in super(BrokenStartRequestsSpider, self).parse(response):
yield req
class SingleRequestSpider(MetaSpider):
seed = None
callback_func = None
errback_func = None
def start_requests(self):
if isinstance(self.seed, Request):
yield self.seed.replace(callback=self.parse, errback=self.on_error)
else:
yield Request(self.seed, callback=self.parse, errback=self.on_error)
def parse(self, response):
self.meta.setdefault('responses', []).append(response)
if callable(self.callback_func):
return self.callback_func(response)
if 'next' in response.meta:
return response.meta['next']
def on_error(self, failure):
self.meta['failure'] = failure
if callable(self.errback_func):
return self.errback_func(failure)
class DuplicateStartRequestsSpider(Spider):
dont_filter = True
name = 'duplicatestartrequests'
distinct_urls = 2
dupe_factor = 3
def start_requests(self):
for i in range(0, self.distinct_urls):
for j in range(0, self.dupe_factor):
url = "http://localhost:8998/echo?headers=1&body=test%d" % i
yield Request(url, dont_filter=self.dont_filter)
def __init__(self, url="http://localhost:8998", *args, **kwargs):
super(DuplicateStartRequestsSpider, self).__init__(*args, **kwargs)
self.visited = 0
def parse(self, response):
self.visited += 1
|