‘’’
京东商品名称价格及评价信息的获取
‘’’
1.########################################################
import re
import time
import csv
import requests
from bs4 import BeautifulSoup
import json
# add headers, download page, check status code, return page
url = 'https://search.jd.com/Search?keyword=%E5%8D%8E%E4%B8%BAp20&enc=utf-8&suggest=1.def.0.V13&wq=%E5%8D%8E%E4%B8%BA&pvid=f47b5d05bba84d9dbfabf983575a6875'
headers = {
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.36 SE 2.X MetaSr 1.0"
}
response = requests.get(url, headers=headers)
response.encoding='utf-8'
print(response.status_code)
# find elements, such as name, item, price, comment, goodrate, comment count
soup_all = BeautifulSoup(response.content, 'lxml')
sp_all_items = soup_all.find_all('li', attrs={'class': 'gl-item'})
for soup in sp_all_items[:1]:
print('-' * 50)
name = soup.find('div', attrs={'class': 'p-name p-name-type-2'}).find('em').text
print('name: ', name)
item = soup.find('div', attrs={'class': 'p-name p-name-type-2'}).find('a')
print('item: ', item['href'], re.search(r'(\d+)', item['href']).group())
price = soup.find_all('div', attrs={'class': 'p-price'})
print('price:', price[0].i.string)
comment = soup.find_all('div', attrs={'class': 'p-commit'})
print('comment url:', comment[0].find('a').attrs['href'])
time.sleep(2)
# need add referer into headers
item_id = re.search(r'(\d+)', item['href']).group()
url = 'https://sclub.jd.com/comment/productPageComments.action?productId=%s&score=0&sortType=5&page=0&pageSize=10&isShadowSku=0&fold=1'
url=url % item_id
print(url,111)
headers = {
"referer": "https://item.jd.com/%s.html" % item_id ,
"User-Agent": "Mozilla/5.0 (Windows NT 10.0; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/58.0.3029.110 Safari/537.36 SE 2.X MetaSr 1.0"
}
print(headers,222)
response = requests.get(url, headers=headers)
response.encoding='gbk'
data = response.json()
print(data)
comment_count = data['productCommentSummary']['commentCount'] #评价人数
print('评价人数:', comment_count)
good_rate = data['productCommentSummary']['goodRate'] #好评率
print('好评率:', good_rate)
pj_content=data['comments']
for i in pj_content:
pj=i['content'] #客户评价
referenceName=i['referenceName'] #客户购买手机款式
print(pj,11111)
#print(referenceName,22222)