Python爬取中国天气网获取全国城市编码并存入MySQL数据库
上代码
import re
import requests
import pymysql
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/67.0.3396.87 Safari/537.36'
}
cites_codes = []
def parse_url(url):
response = requests.get(url, headers=headers)
text = response.content.decode('utf-8')
info = re.findall(
r'<div class="conMidtab">.*?<div class="conMidtab" style="display:none;">',text,re.DOTALL)[0]
infos = re.findall(
r'<td width="83" height="23".*?<a .*?weather/(.*?)\.s.*?>(.*?)</a>', info, re.DOTALL)
for i in infos:
city_code = [i[1],i[0]]
cites_codes.append(city_code)
def store_2_mysql():
conn = pymysql.connect(host='localhost', user='root',
password='712688', database='test_demo', port=3306)
cursor = conn.cursor()
for i in cites_codes:
sql = '''
insert into city_code values(0,'%s','%s')''' % (i[0], i[1])
cursor.execute(sql)
conn.commit()
conn.close()
def main():
base_url = 'http://www.weather.com.cn/textFC/{}.shtml'
cites = ['hb', 'db', 'hd', 'hz', 'hn', 'xb', 'xn', 'gat']
for i in cites:
url = base_url.format(i)
parse_url(url)
store_2_mysql()
if __name__ == '__main__':
main()