Python爬虫抓取气象_bs4+定时器+mysql+对象_一蓑烟雨任平生

最新推荐文章于 2024-07-01 08:47:12 发布

一蓑烟雨任平生√

最新推荐文章于 2024-07-01 08:47:12 发布

阅读量308

点赞数 2

分类专栏： python 爬虫

本文链接：https://blog.csdn.net/Jaeger_Java/article/details/115462980

版权

Python爬虫定时任务 BeautifulSoup MySQL 面向对象

关键词由CSDN通过智能技术生成

python 同时被 2 个专栏收录

47 篇文章 4 订阅

订阅专栏

爬虫

35 篇文章 3 订阅

订阅专栏

前言

麻雀虽小五脏俱全这篇爬虫文章涉及的技术不少

bs4抓取数据 (之前一直用xpath感觉一种东西吃多了会腻)
定时器(一次执行终身执行懒人必备)
mysql(数据库存数据的地方)
对象(面向对象编程)

说啥呢?直接扔代码吧

看不懂的话你细品留言也可以进群也可以

# -*- coding: utf-8 -*-
"""
# @Time : 2021/4/6 10:10 

# @Author : 一蓑烟雨任平生

# @File : 天气.py 

# @Software: PyCharm
"""
import datetime
import threading
from datetime import date

import pymysql
import requests
from bs4 import BeautifulSoup

conn = pymysql.connect(host='127.0.0.1', user='root', passwd='123456', db='feifei', charset='utf8')
cur = conn.cursor()
print("数据库已连接")


class ZhangZhang:
    def __init__(self):
        self.head = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:87.0) Gecko/20100101 Firefox/87.0'}
        self.babaUrl = 'http://www.agri.cn/qxny/nqyw'

    def getTime(self):
        # 获取现在时间
        now_time = datetime.datetime.now()
        # 获取明天时间
        next_time = now_time + datetime.timedelta(days=+1)
        next_year = next_time.date().year
        next_month = next_time.date().month
        next_day = next_time.date().day
        # 获取明天3点时间
        next_time = datetime.datetime.strptime(
            str(next_year) + "-" + str(next_month) + "-" + str(next_day) + " 03:00:00",
            "%Y-%m-%d %H:%M:%S")
        timer_start_time = (next_time - now_time).total_seconds()
        print(timer_start_time)
        return timer_start_time;

    def get_all_url(self):
        ee = []
        page_one = self.bs_Url('http://www.agri.cn/qxny/nqyw/index.htm')
        bb = [page_one.find('td', class_='bk_7').find_all("table")[i] for i in range(1, 8, 2)]
        time4 = date.today() - datetime.timedelta(days=4)
        for i in bb:
            cc = i.find("td", class_='hui_14').text[1:-1]
            if str(time4) == cc:
                dd = i.find('a', class_='link03')['href'][1:]
                ee.append(dd)
        ff = [self.babaUrl + fp for fp in ee]
        return ff

    def get_info(self, list):
        for i in list:
            pic = i.split('/t')[0] + '/'
            shuju = self.bs_Url(i)
            title = shuju.find('td', class_='hui_15_cu').text
            publish_time = shuju.find('td', class_='hui_12-12').text.split("：")[1][:10]
            source = shuju.find('td', class_='hui_12-12').text.split("：")[3]
            content = str(shuju.find('div', class_='TRS_Editor'))
            content = content.replace("./", pic)
            sql = "insert into weather_info(title,publish_time,source,content) VALUES (%s,%s,%s,%s)"
            cur.execute(sql, (title, publish_time, source, content))
        conn.commit()
        cur.close()
        conn.close()

    def bs_Url(self, wangzhi):
        yy = requests.get(wangzhi, self.head).content
        page_info = BeautifulSoup(yy, "html.parser")
        return page_info

    def run(self):
        conn.ping(reconnect=True)
        print("------爬虫程序开始------")
        # 获取今日Url然后详情
        self.get_info(self.get_all_url());
        # 获取信息
        timer_start_time = self.getTime();
        timer = threading.Timer(timer_start_time, self.run)
        timer.start()


if __name__ == '__main__':
    ZhangZhang().run()