防爬虫网站用 selenium破解

最新推荐文章于 2024-05-12 19:41:06 发布

tomandmath

最新推荐文章于 2024-05-12 19:41:06 发布

阅读量287

点赞数

分类专栏： python

本文链接：https://blog.csdn.net/qq_42676042/article/details/106939185

版权

python 专栏收录该内容

4 篇文章 0 订阅

订阅专栏

#!/usr/bin/env python  获取单个信息
# coding=utf-8
import datetime
from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException
import selenium.webdriver.support.ui as ui

browser = webdriver.Firefox()

def is_visible(locator, timeout = 10):
    try:
        ui.WebDriverWait(browser, timeout).until(EC.visibility_of_element_located((By.XPATH, locator)))
        return True
    except TimeoutException:
        return False
    
browser.get("http://zssom.sysu.edu.cn/zh-hans/teacher/377")
is_visible('/html/body/div[2]/div[2]/div[1]')
html = browser.page_source
content = BeautifulSoup(html, "lxml")
description = content.find(attrs={"name":"description"})['content']
print(description)

#以下还有问题 请小心 无法运行
#以下为中山医实例 此处向中山大学致以崇高的敬意与祝福 欢迎提意见
#!/usr/bin/env python 如有侵权 亲联系 删文章哦
# coding=utf-8
import datetime
from bs4 import BeautifulSoup
from selenium import webdriver
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.common.by import By
from selenium.webdriver.support import expected_conditions as EC
from selenium.common.exceptions import TimeoutException
import selenium.webdriver.support.ui as ui

browser = webdriver.Firefox()

def is_visible(locator, timeout = 10):
    try:
        ui.WebDriverWait(browser, timeout).until(EC.visibility_of_element_located((By.XPATH, locator)))
        return True
    except TimeoutException:
        return False
#编辑 url
i =350
while i <381:
    i=i+1
    a = str(i)
    b = "http://zssom.sysu.edu.cn/zh-hans/teacher/"
    url= b+a
    browser.get(url)
    
    if is_visible('/html/body/div[2]/div[2]/div[1]'):
        html = browser.page_source
        content = BeautifulSoup(html, "lxml")
        #获取老师信息内容
        description = content.find(attrs={"name":"description"})['content']
        print(description)
    else:
        print("获取内容为空")
else:
    print("over")

tips
1.用selenium 模拟浏览器行为
2.拼接字符串设计url
3.bs4 获取节点数据

tomandmath

关注

0
点赞
踩
2

收藏

觉得还不错? 一键收藏
0
评论
防爬虫网站用 selenium破解

#以下为中山医实例此处向中山大学致以崇高的敬意与祝福欢迎提意见#!/usr/bin/env python 如有侵权亲联系删文章哦# coding=utf-8import datetimefrom bs4 import BeautifulSoupfrom selenium import webdriverfrom selenium.webdriver.support.ui import WebDriverWaitfrom selenium.webdriver.common.by impo
复制链接

扫一扫