(爬虫)
**
百度图片小爬虫
**
完整的代码
运行前要安装好requests库,运行程序后,输入要爬取图片的名称,就能自动到百度图片上爬取需要的图片。
import requests
import re
pic=input('你想从百度图片里下载什么图片???\n')
url="https://image.baidu.com/search/index?tn=baiduimage&ipn=r&ct=201326592&cl=2&lm=-1&st=-1&fm=result&fr=&sf=1&fmq=1597996649625_R&pv=&ic=&nc=1&z=&hd=&latest=©right=&se=1&showtab=0&fb=0&width=&height=&face=0&istype=2&ie=utf-8&hs=2&sid=&word=%E7%8C%AB"
#百度图片的网址
headers={
"Cookie":"BAIDUID=D59D9FB8269C678FBFA350D513329907:FG=1; BIDUPSID=DC2C0574D794309CACA49DF084700264; BDORZ=B490B5EBF6F3CD402E515D22BCDA1598; PSTM=1598196589; BDUSS=llXSDVyaTRYUmF3dUJNenJ2MFRrUX5XTS15Q2dUWXI4QTdUVENrfm5vN343MnRmRVFBQUFBJCQAAAAAAAAAAAEAAAB6mI0Zdru509C1sLjiAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAP9iRF~~YkRfW; BDUSS_BFESS=llXSDVyaTRYUmF3dUJNenJ2MFRrUX5XTS15Q2dUWXI4QTdUVENrfm5vN343MnRmRVFBQUFBJCQAAAAAAAAAAAEAAAB6mI0Zdru509C1sLjiAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAP9iRF~~YkRfW; BDRCVFR[feWj1Vr5u3D]=I67x6TjHwwYf0; delPer=0; PSINO=7; H_PS_PSSID=32668_1462_32536_31254_32046_32116_32618_32500_32481; BDRCVFR[X_XKQks0S63]=mk3SLVN4HKm; userFrom=www.baidu.com; BDRCVFR[-pGxjrCMryR]=mk3SLVN4HKm; firstShowTip=1; BDRCVFR[dG2JNJb_ajR]=mk3SLVN4HKm",
"User-Agent":"Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/78.0.3904.108 Safari/537.36",
"Referer":"https://image.baidu.com/"
}
#请求头,
kw={
"word":pic
}
#要爬取的图片名称
res=requests.get(url,headers=headers,params=kw)#res是一个response对象
content=res.content.decode('utf8')#content解码为比特流,decode指定编码格式为utf-8
detail_urls=re.findall('"objURL":"(.*?)"',content,re.DOTALL)
#通过查看源代码发现图片网址对应objURL。得到图片地址
j=0
for i in detail_urls:
res=requests.get(i)
#得到图片相应
content=res.content
#得到图片内容(bytes)
print(i)
if i[-3:]=='jpg':#判断图片格式
with open('{}{}.jpg'.format(pic,j),'wb') as f:
f.write(content)
elif i[-4:]=='jpeg':
with open('{}{}.jpeg'.format(pic,j),'wb') as f:
f.write(content)
elif i[-3:]=='png':
with open('{}{}.png'.format(pic,j),'wb') as f:
f.write(content)
else:
continue
j+=1
#下载图片