今天,因为某种需要,要对国外大学排行榜进行数据的爬取。所以,对那个网站的一些数据进行的了爬取。
对爬取到的数据进行存储到mysql数据库中。
网站地址:点击打开链接
# _._ coding:utf-8 _._#
import lxml
from lxml import etree
import requests
import MySQLdb
# 打开数据库连接
db = MySQLdb.connect("xxxx","xxxx","xxxx","xxxx" )
#设置数据库编码
db.set_character_set('utf8')
# 使用cursor()方法获取操作游标
cursor = db.cursor()
headers = {
'User-Agent' : 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/62.0.3202.94 Safari/537.36'
}
r = requests.get(url='https://www.usnews.com/education/best-global-universities/search?region=&subject=&name=',headers=headers)
html = r.text.encode('utf-8')
# result = etree.tostring(html, pretty_print=True)
# print result
result = etree.HTML(html)
# # print result
# seq1 = []
allPage = result.xpath('//div[@class="pagination"]//a[last()-1]')
sumPage = allPage[0].text
subjects_list = []
select = result.xpath('//select[@name="subject"]')
allOptions = select[0].xpat