python爬去知乎动态内容_如何利用python 爬取知乎上面的数据

最新推荐文章于 2020-11-26 05:37:31 发布

weixin_39742727

最新推荐文章于 2020-11-26 05:37:31 发布

阅读量137

点赞数

文章标签： python爬去知乎动态内容

展开全部

^#!/usr/bin/env python

# -*- coding: utf-8 -*-

# @Author: Administrator

# @Date: 2015-10-31 15:45:27

# @Last Modified by: Administrator

# @Last Modified time: 2015-11-23 16:57:31

import requests

import sys

import json

import re

reload(sys)

sys.setdefaultencoding('utf-8')

#获取到匹配62616964757a686964616fe59b9ee7ad9431333339663965字符的字符串

def find(pattern,test):

finder = re.search(pattern, test)

start = finder.start()

end = finder.end()

return test[start:end-1]

cookies = {

'_ga':'GA1.2.10sdfsdfsdf', '_za':'8d570b05-b0b1-4c96-a441-faddff34',

'q_c1':'23ddd234234',

'_xsrf':'234id':'"ZTE3NWY2ZTsdfsdfsdfWM2YzYxZmE=|1446435757|15fef3b84e044c122ee0fe8959e606827d333134"',

'z_c0':'"QUFBQXhWNGZsdfsdRvWGxaeVRDMDRRVDJmSzJFN1JLVUJUT1VYaEtZYS13PT0=|14464e234767|57db366f67cc107a05f1dc8237af24b865573cbe5"',

'__utmt':'1', '__utma':'51854390.109883802f8.1417518721.1447917637.144c7922009.4',

'__utmb':'518542340.4.10.1447922009', '__utmc':'51123390', '__utmz':'5185435454sdf06.1.1.utmcsr=zhihu.com|utmcgcn=(referral)|utmcmd=referral|utmcct=/',

'__utmv':'51854340.1d200-1|2=registration_date=2028=1^3=entry_date=201330318=1'}

headers = {'user-agent':

'Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/38.0.2125.111 Safari/537.36',

'referer':'http://www.zhihu.com/question/following',

'host':'www.zhihu.com','Origin':'http://www.zhihu.com',

'Content-Type':'application/x-www-form-urlencoded; charset=UTF-8',

'Connection':'keep-alive','X-Requested-With':'XMLHttpRequest','Content-Length':'81',

'Accept-Encoding':'gzip,deflate','Accept-Language':'zh-CN,zh;q=0.8','Connection':'keep-alive'

}

#多次访问之后，其实一加载时加载20个问题，具体参数传输就是offset，以20递增

dicc = {"offset":60}

n=20

b=0

# 与爬取图片相同的是，往下拉的时候也会发送http请求返回json数据，但是不同的是，像模拟登录首页不同的是除了

# 发送form表单的那些东西后，知乎是拒绝了我的请求了，刚开始以为是headers上的拦截，往headers添加浏览器

# 访问是的headers那些信息添加上，发现还是拒绝访问。

#想了一下，应该是cookie原因。这个加载的请求和模拟登录首页不同

#所以补上其他的cookies信息，再次请求，请求成功。

for x in xrange(20,460,20):

n = n+20

b = b+20

dicc['offset'] = x

formdata = {'method':'next','params':'{"offset":20}','_xsrf':'20770d88051f0f45e941570645f5e2e6'}

#传输需要json串，和python的字典是有区别的，需要转换

formdata['params'] = json.dumps(dicc)

# print json.dumps(dicc)

# print dicc

circle = requests.post("http://www.zhihu.com/node/ProfileFollowedQuestionsV2",

cookies=cookies,data=formdata,headers=headers)

#response内容其实爬过一次之后就大同小异了。都是

#问题返回的json串格式

# {"r":0,

# "msg": ["

# \n

205K<\/div>\n

\u6d4f\u89c8<\/div>\n

# <\/span>\n

\n

# \u4ec0\u4e48\u4fc3\u4f7f\u4f60\u8d70\u4e0a\u72ec\u7acb\u5f00\u53d1\u8005\u4e4b\u8def\uff1f<\/a>\n

# <\/h2>\n

# href=\"javascript:;\" id=\"sfb-868760\">

# "

# \n

157K<\/div>\n

\u6d4f\u89c8<\/div>\n

# <\/span>\n

\n

# \u672c\u79d1\u6e23\u6821\u7684\u5b66\u751f\u5982\u4f55\u8fdb\u5165\u7f8e\u5e1d\u725b\u6821\u8bfbPhD\uff1f<\/a>\n

# <\/h2>\n

# <\/span>\n112 \u4e2a\u56de\u7b54\n•<\/span>\n1582 \u4eba\u5173\u6ce8\n

# <\/div>\n<\/div>\n<\/div>"]}

# print circle.content

#同样json串需要自己转换成字典后使用

jsondict = json.loads(circle.text)

msgstr = jsondict['msg']

# print len(msgstr)

#根据自己所需要的提取信息规则写出正则表达式

pattern = 'question\/.*?/a>'

try:

for y in xrange(0,20):

wholequestion = find(pattern, msgstr[y])

pattern2 = '>.*?<'

finalquestion = find(pattern2, wholequestion).replace('>','')

print str(b+y)+" "+finalquestion

#当问题已经访问完后再传参数抛出异常此时退出循环

except Exception, e:

print "全部%s个问题" %(b+y)

weixin_39742727

关注

0
点赞
踩
0

收藏

觉得还不错? 一键收藏
0
评论
python爬去知乎动态内容_如何利用python 爬取知乎上面的数据

展开全部^#!/usr/bin/env python# -*- coding: utf-8 -*-# @Author: Administrator# @Date: 2015-10-31 15:45:27# @Last Modified by: Administrator# @Last Modified time: 2015-11-23 16:57:31import requestsimpo...
复制链接

扫一扫