python lxml.html Beautifulsoup 打开本地html 文件方法

最新推荐文章于 2025-04-02 14:19:49 发布

tianyue100

最新推荐文章于 2025-04-02 14:19:49 发布

阅读量4.6k

点赞数 3

python 打开本地html 文件方法,lxml BS4 open local .html file

REF:

https://docs.python-guide.org/scenarios/scrape/

https://www.datacamp.com/community/tutorials/python-xml-elementtree

1 lxml.html

from lxml import html,etree
import requests

from bs4 import BeautifulSoup

"""
#-----------------------use URL-----------------------------------
base_url = "http://www.runoob.com/"
page = requests.get(base_url)
#page = requests.get('http://econpy.pythonanywhere.com/ex/001.html')
tree = html.fromstring(page.content)

#-----------------------use URL end-----------------------------------
"""


file ='./runoob_cainiao.html'
tree = html.parse(file)

#This will create a list of buyers:

lists = root.xpath('/html/body/div[4]/div/div[2]/div/a/strong/text()')

#print the lists
for list in lists:
    print('list: '+ '\n' ,list)

2 Beautifulsoup

REF https://blog.csdn.net/fwj_ntu/article/details/78843872

from lxml import html,etree
import requests

from bs4 import BeautifulSoup

"""
#-----------------------use URL-----------------------------------
base_url = "http://www.runoob.com/"
page = requests.get(base_url)
#page = requests.get('http://econpy.pythonanywhere.com/ex/001.html')
tree = html.fromstring(page.content)

#-----------------------use URL end-----------------------------------
"""


file ='./runoob_cainiao.html'
#tree = html.parse(file)
#使用open函数打开文件
htmlfile = open(file, 'r', encoding='utf-8')

#读取html的句柄内容
htmlhandle = htmlfile.read()

#使用Beautifulsoup解析
soup = BeautifulSoup(htmlhandle, features='lxml')


print(soup.find('h1').get_text(), 'url: ',file)