python爬虫
lxml解析html代码和文件
- 先将html代码解析为html文档
- 按字符串序列化
- 成功解析
# -*- codeing = utf-8 -*-
# coding=gbk
# @Autor:ggui20
from lxml import etree
text = '''
<div>
<ul>
<li class="item-0"><a href="link1.html">first item</a></li>
<li class="item-1"><a href="link2.html">second item</a></li>
<li class="item-inactive"><a href="link3.html">third item</a></li>
<li class="item-1"><a href="link4.html">fourth item</a></li>
<li class="item-0"><a href="link5.html">fifth item</a>
</ul>
</div>
'''
# 将字符串解析为html文档
html = etree.HTML(text)
print(html)
# 按字符串序列化html
result = etree.tostring(html).decode('utf-8')
print(result)
# 读取
html = etree.parse('hello.html')
result = etree.tostring(html).decode('utf-8')
print(result)