pyth 数据采集

最新推荐文章于 2023-04-23 16:15:00 发布

元永真

最新推荐文章于 2023-04-23 16:15:00 发布

阅读量381

点赞数

分类专栏： paython学习总结

本文链接：https://blog.csdn.net/weixin_37855495/article/details/66978809

版权

paython学习总结专栏收录该内容

16 篇文章 0 订阅

订阅专栏

数据采集

=============

beautifulsoup4文档：

https://www.crummy.com/software/BeautifulSoup/bs4/doc/

安装beautifulsoup4

https://www.crummy.com/software/BeautifulSoup/
pip install beautifulsoup4
from bs4 import BeautifulSoup

===================
from urllib.request import urlopen
C:\Users\ljf>pip install beautifulsoup4 //安装 bs4
from bs4 import BeautifulSoup
PyDev - http://pydev.org/nightly
http://www.ibm.com/developerworks/cn/opensource/os-cn-ecl-pydev/

import urllib.request
#print (urllib.request)
class HtmlDownloader(object):

def download(self,url):
if url is None:
return None

response = urllib.request.urlopen(url)

if response.getcode() != 200:
return None
return response.read()
#obj = HtmlDownloader()
#obj.download("http://baike.bai du.com/view/21087.htm")

from bs4 import BeautifulSoup
import re
#import urlparse
#import urllib.request

class HtmlParser(object):

def _get_new_urls(self,page_url,soup):
new_urls = set()
links = soup.find_all("a",href=re.compile(r"/view/\d+\.htm"))
for link in links:
new_url = link['href']
# print(new_url)
new_full_url = "http://baike.baidu.com"+new_url
# print (new_full_url)
new_urls.add(new_full_url)

return new_urls

def _get_new_data(self,page_url,soup):
res_data = {}
res_data['url'] = page_url
ttile_node = soup.find('dd',class_="lemmaWgt-lemmaTitle-title").find("h1")

res_data['title'] = ttile_node.get_text()

summary_node = soup.find("div",class_="lemma-summary")
res_data['summary'] = summary_node.get_text()
return res_data

def parse(self,page_url,html_cont):
if page_url is None or html_cont is None:
return

soup = BeautifulSoup(html_cont,'html.parser',from_encoding = "utf-8")
# print (soup)
new_urls = self._get_new_urls(page_url,soup)
# print (new_urls)
new_data = self._get_new_data(page_url,soup)
return new_urls, new_data

#test = HtmlParser()
#response = urllib.request.urlopen("http://baike.baidu.com/view/21087.htm")
#test.parse("http://baike.baidu.com/view/21087.htm",response.read())

import sys
import os
curPath = os.path.abspath(os.path.dirname(__file__))
rootPath = os.path.split(curPath)[0]
sys.path.append(rootPath)
from baike_spider import url_manager,html_downloader, html_parser,html_outputer

class SpiderMain(object):
def __init__(self):
print ('22222222222')
# url 管理器，模板下载器，下载数据解析器，解析数据输入器。模块调度器
self.urls = url_manager.UrlManager()
self.downloader = html_downloader.HtmlDownloader()
self.parser = html_parser.HtmlParser()
self.outputer = html_outputer.HtmlOutputer()

def craw(self,root_url):
count = 1
self.urls.add_new_url(root_url)

while self.urls.has_new_url():
try:
new_url = self.urls.get_new_url()
print (count,new_url)
# print craw (‘ %d : %s‘) % (count,new_url)
html_cont = self.downloader.download(new_url)
new_urls,new_data = self.parser.parse(new_url,html_cont)
print (new_urls)
print (new_data)
self.urls.add_new_urls(new_urls)
self.outputer.collect_data(new_data)
if count == 10:
break

count = count + 1

except:
print ('craw fail')

self.outputer.output_html()

if __name__ =="__main__":
print ('1111111111111')
root_url = "http://baike.baidu.com/view/21087.htm"
obj_spider = SpiderMain()
obj_spider.craw(root_url)

class UrlManager(object):
def __init__(self):
self.new_urls = set()
self.old_urls = set()

def add_new_url(self,url):
if url is None:
return
if url not in self.new_urls and url not in self.old_urls:
self.new_urls.add(url)

def add_new_urls(self,urls):
if urls is None or len(urls) == 0:
return
for url in urls:
self.add_new_url(url)

def has_new_url(self):
return len(self.new_urls) != 0

def get_new_url(self):
new_url = self.new_urls.pop()
self.old_urls.add(new_url)
return new_url

元永真

关注

0
点赞
踩
0

收藏

觉得还不错? 一键收藏
0
评论
pyth 数据采集

1import urllib.request#print (urllib.request)class HtmlDownloader(object): def download(self,url): if url is None: return None response = urllib.requ
复制链接

扫一扫

专栏目录