python批量检索文献_利用requests-html在python3.7下批量下载论文

#-*- coding:utf-8 -*-

# python3.7 requests-html2.0 pycharm

import requests_html

import requests

import random

import sys

HOMEURL = "http://www.padsweb.rwth-aachen.de/wvdaalst/publications/"

session = requests_html.Session()

list_url = 'http://www.padsweb.rwth-aachen.de/wvdaalst/publications/publications.html'

USER_AGENTS = [

"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_7_3) AppleWebKit/535.20 (KHTML, like Gecko) Chrome/19.0.1036.7 Safari/535.20",

"Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.71 Safari/537.1 LBBROWSER",

"Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/535.11 (KHTML, like Gecko) Chrome/17.0.963.84 Safari/535.11 LBBROWSER",

"Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E)",

"Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E)",

"Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 5.1; Trident/4.0; SV1; QQDownload 732; .NET4.0C; .NET4.0E; 360SE)",

"Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E)",

"Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.89 Safari/537.1",

"Mozilla/5.0 (iPad; U; CPU OS 4_2_1 like Mac OS X; zh-cn) AppleWebKit/533.17.9 (KHTML, like Gecko) Version/5.0.2 Mobile/8C148 Safari/6533.18.5",

"Mozilla/5.0 (Windows NT 6.1; Win64; x64; rv:2.0b13pre) Gecko/20110307 Firefox/4.0b13pre",

"Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:16.0) Gecko/20100101 Firefox/16.0",

"Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.11 (KHTML, like Gecko) Chrome/23.0.1271.64 Safari/537.11",

"Mozilla/5.0 (X11; U; Linux x86_64; zh-CN; rv:1.9.2.10) Gecko/20100922 Ubuntu/10.10 (maverick) Firefox/3.6.10"

]

# 获取页面中所有可以下载的论文名及链接中的名字

# 如文件名Timed coloured Petri nets and their application to logistics.pdf

# 链接中的名字:publicationsp7.pdf file

def get_list(url):

response = session.get(url)

all_link = response.html.find('#main a') # 获取页面所有图书详情链接

# Timed coloured Petri nets and their application to logistics.

for link in all_link:

# for i in range(1,30):

# link=all_link[i]

href = link.attrs['href']

hrefText = link.text

if '.pdf' in href:

# 处理含有多个'/'的情况,取最后一个斜杠后的部分

pdfName = str(hrefText).split('/')[-1] + ".pdf"

if '?' in pdfName:

pdfName = pdfName.replace('?', '')

url = HOMEURL + str(href)

download(pdfName, url)

# 下载论文

def download(pdfName, url):

# 随机浏览器 User-Agent

headers = {"User-Agent": random.choice(USER_AGENTS)}

# 获取文件名

filename = url.split('/')[-1]

print("filename:", filename, "fileName:", pdfName)

# 如果 url 里包含 .pdf

if ".pdf" in url:

file = 'd://books/' + pdfName # 文件路径写死了,运行时当前目录必须有名 book 的文件夹

try:

with open(file, 'wb') as f:

print("正在下载 %s" % filename)

response = requests.get(url, stream=True, headers=headers)

# 获取文件大小

total_length = response.headers.get('content-length')

# print("链接:",url)

# print('response.content:',response.content)

# print(requests.get(url))

# 如果文件大小不存在,则直接写入返回的文本

if total_length is None:

f.write(response.content)

# print("文件为空:",pdfName,"\t",url,'\t',response.content)

else:

# 下载进度条

dl = 0

total_length = int(total_length) # 文件大小

for data in response.iter_content(chunk_size=4096): # 每次响应获取 4096 字节

dl += len(data)

f.write(data)

done = int(50 * dl / total_length)

sys.stdout.write("\r[%s%s]" % ('=' * done, ' ' * (50 - done))) # 打印进度条

sys.stdout.flush()

except IOError as e:

print(e)

else:

print(filename + '下载完成!')

if __name__ == '__main__':

get_list(list_url)

安装requests-html过程中出现的问题

1、cannot import name 'HTMLSession' from 'requests_html' (C:\Users\owlish\AppData\Local\Programs\Python\Python37\lib\site-packages\requests_html.py)

很尴尬,我的文件名叫“requests.py”,和关键字冲突,改个名就行了。

2、pycharm安装后找不到本地已经安装的模块

035b0327321e

创建虚拟环境

035b0327321e

继承全局包

如图,创建了一个虚拟环境就行了,并勾选“Inherit global site-packages”。仿佛是因为安装目录在c盘,权限不够。

  • 0
    点赞
  • 0
    收藏
    觉得还不错? 一键收藏
  • 0
    评论

“相关推荐”对你有帮助么?

  • 非常没帮助
  • 没帮助
  • 一般
  • 有帮助
  • 非常有帮助
提交
评论
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值