|
楼主 |
发表于 2019-4-18 10:52:52
|
显示全部楼层
- import re
- import urllib.request
- import random
- def urlopen(url):
- req = urllib.request.Request(url)
- req.add_header('User-Agent','Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/63.0.3239.26 Safari/537.36 Core/1.63.6823.400 QQBrowser/10.3.3117.400')
- proxies = ['124.152.32.140:53281', '163.125.157.53:8888', '163.125.157.49:8888']
- proxy = random.choice(proxies)
- proxy_support = urllib.request.ProxyHandler({'http':proxy})
- opener = urllib.request.build_opener(proxy_support)
- urllib.request.install_opener(opener)
- html = urllib.request.urlopen(req)
- html = html.read()
- return html
- def zhuyao(url):
- a = urlopen(url)
- a = a.decode('utf-8')
- p = r'<img src="([^"]+?\.jpg)"'
- tplj = re.findall(p,a)
- for each in tplj:
- b = urlopen(each)
- with open('一张妹子图','wb') as f:
- f.write(b)
- if __name__=='__main__':
- url = 'https://www.mzitu.com/176550/6'
- zhuyao(url)
复制代码 改成这样好像也不行啊
Traceback (most recent call last):
File "C:\Users\tx\Desktop\妹子图片爬取 - 副本.py", line 31, in <module>
zhuyao(url)
File "C:\Users\tx\Desktop\妹子图片爬取 - 副本.py", line 24, in zhuyao
b = urlopen(each)
File "C:\Users\tx\Desktop\妹子图片爬取 - 副本.py", line 15, in urlopen
html = urllib.request.urlopen(req)
File "C:\Python3.7.2\lib\urllib\request.py", line 222, in urlopen
return opener.open(url, data, timeout)
File "C:\Python3.7.2\lib\urllib\request.py", line 531, in open
response = meth(req, response)
File "C:\Python3.7.2\lib\urllib\request.py", line 641, in http_response
'http', request, response, code, msg, hdrs)
File "C:\Python3.7.2\lib\urllib\request.py", line 569, in error
return self._call_chain(*args)
File "C:\Python3.7.2\lib\urllib\request.py", line 503, in _call_chain
result = func(*args)
File "C:\Python3.7.2\lib\urllib\request.py", line 649, in http_error_default
raise HTTPError(req.full_url, code, msg, hdrs, fp)
urllib.error.HTTPError: HTTP Error 403: Forbidden
>>> |
|