import urllib.request
import re
import os
def url_open(url):
req=urllib.request.Request(url)
req.add_header('User-Agent','Mozilla/5.0 (Windows NT 6.1; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/60.0.3112.113 Safari/537.36')
response=urllib.request.urlopen(url)
html=response.read().decode('utf-8')
return html
def get_img(html):
p=r'<img src="([^"]+\.jpg)"'
imglist=re.findall(p,html)
'''
for each in imglist:
print (each)
'''
for each in imglist:
filename=each.split("/")[-1]
urllib.request.urlretrieve(each,filename,None)
if __name__=='__main__':
os.mkdir("E:\Pict")
os.chdir("E:\Pict")
url='https://www.zhihu.com/question/40007169'
get_img(url_open(url))
第一个小爬虫--爬取图片并保存
最新推荐文章于 2023-04-21 11:30:33 发布