爬取数据之Chrome

最新推荐文章于 2024-07-22 14:55:58 发布

upc-a-x

最新推荐文章于 2024-07-22 14:55:58 发布

阅读量81

点赞数

文章标签： chrome python 前端

本文链接：https://blog.csdn.net/MrANdreamer/article/details/130043636

版权

from selenium import webdriver
from selenium.webdriver.common.by import By
from selenium.webdriver.support.ui import WebDriverWait
from selenium.webdriver.support import expected_conditions as EC
from selenium.webdriver.common.by import By
from selenium.webdriver.chrome.service import Service
from bs4 import BeautifulSoup

driver =webdriver.Chrome()
url='https://www.kylc.com/stats/global/yearly/g_gdp/1960.html'
xpath="/html/body/div[2]/div[1]/div[5]/div[1]/div/div/div/table"
driver.get(url)

table1 = driver.find_element(By.XPATH,xpath).get_attribute('innerHTML')
soup=BeautifulSoup(table1,"html.parser")
table = soup.find_all('tr')
for row in table:
    cols=[col.text for col in row.find_all('td')]
    if len(cols)==0 or not cols[0].isdigit():
        continue

    print(cols)
#print(table1)