这里只是简单的把image_list里面的image_url提取出来,具体的结构化编程请自行编写。
import requests
from urllib.parse import urlencode
params = {
'aid': 24,
'app_name': 'web_search',
'offset': 0,
'format': 'json',
'keyword': '街拍',
'autoload': 'true',
'count': 20,
'en_qc': 1,
'cur_tab': 1,
'from': 'search_tab',
'pd': 'synthesis',
'timestamp': 1583996635550,
'_signature': 'OeQwkAAgEBD.s4kdPmRnbTnlcYAAGeU8wq5tsZ-LY-cgKJxWpkh1SR0knLZinWhBvThA9ExwTlkT5e53XdDTaQ.J7sF6IObokwGfv5IFtDIp8xg5ZxvhRdt424k7BL.1trF',
}
headers = {
'cookie': 'tt_webid=6788065855508612621; WEATHER_CITY=%E5%8C%97%E4%BA%AC; tt_webid=6788065855508612621; csrftoken=495ae3a5659fcdbdb78e255464317789; s_v_web_id=k66hcay0_qsRG7emW_x2Qj_4R3o_AeAG_iT4JWmz83jzr; __tasessionId=23dn3qk0f1580738708512',
'referer': 'https://www.toutiao.com/search/?keyword=%E8%A1%97%E6%8B%8D',
'user-agent': 'Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/69.0.3497.100 Safari/537.36',
'x-requested-with': 'XMLHttpRequest'
}
url = 'https://www.toutiao.com/api/search/content/?' + urlencode(params)
print(url)
response = requests.get(url, headers=headers).json()
x = response['data']
for i in x:
try:
image_list = i['image_list']
for i in image_list:
print(i['url'])
except KeyError as e:
print('pass')