import requests
import re
import os
image = '表情包'
if not os.path.exists(image):
os.mkdir(image)
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:98.0) Gecko/20100101 Firefox/98.0'
}
for page in range(41,47):
response = requests.get(f'https://qq.yh31.com/zjbq/List_{page}.html',headers=headers)
#https://qq.yh31.com/zjbq/List_41.html
response.encoding = 'GBK'
response.encoding = 'utf-8'
print(response.request.headers)
print(response.status_code)
t = '<img src="(.*?)" alt="(.*?)" width="160" height="120">'#正则表达式
result = re.findall(t, response.text)
print(result)
for img in result:
print(img)
res = requests.get(img[0])
print(res.status_code)
s = img[0].split('.')[-1] #//截取地址末尾,得到表情包格式,如.jpg .git
with open(image + '/' + img[1] + '.' + s, mode='wb') as file:
file.write(res.content)
正则表达式就可以爬取到你想要的表情包啦