|
|

楼主 |
发表于 2018-4-20 14:24:38
|
显示全部楼层
哥哥,是这样写的嘛,还是不行啊
import urllib.request
import re
def url_open(url):
req=urllib.request.Request(url)
req.add_headers={
'Accept': 'image/webp,image/apng,image/*,*/*;q=0.8','Accept-Encoding': 'gzip, deflate','Accept-Language': 'zh-CN,zh;q=0.9','Cache-Control': 'no-cache','Connection': 'keep-alive',
'DNT': '1','Host': 'i1.umei.cc','Pragma': 'no-cache','Referer': 'http://www.umei.cc/meinvtupian/waiguomeinv/hanguomeinv.htm','User-Agent': 'Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/66.0.3359.117 Safari/537.36'}
response = urllib.request.urlopen(req)
html = response.read().decode('utf-8')
print(html)
return html
def get_img(html):
p=r'<img src="([^"]+\.jpg)"'
imglist = re.findall(p,html)
for each in imglist:
filename = each.split('/')[-1]
print(filename)
print(each)
urllib.request.urlretrieve(each,filename,None)
if __name__ == '__main__':
url="http://www.umei.cc/meinvtupian/waiguomeinv/hanguomeinv.htm"
get_img(url_open(url)) |
|