BeautifulSoup库学习笔记

import requests
from bs4 import BeautifulSoup
import lxml
# data = requests.get('https://book.douban.com/').text

data = '''
<ul>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-main","uid":"0"}' href="https://www.douban.com" target="_blank">豆瓣</a></li>
<li class="on"><a data-moreurl-dict='{"from":"top-nav-click-book","uid":"0"}' href="https://book.douban.com">读书</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-movie","uid":"0"}' href="https://movie.douban.com" target="_blank">电影</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-music","uid":"0"}' href="https://music.douban.com" target="_blank">音乐</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-location","uid":"0"}' href="https://www.douban.com/location" target="_blank">同城</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-group","uid":"0"}' href="https://www.douban.com/group" target="_blank">小组</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-read","uid":"0"}' href="https://read.douban.com/?dcs=top-nav&amp;dcm=douban" target="_blank">阅读</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-fm","uid":"0"}' href="https://douban.fm/?from_=shire_top_nav" target="_blank">FM</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-time","uid":"0"}' href="https://time.douban.com/?dt_time_source=douban-web_top_nav" target="_blank">时间</a>
</li>
<li class=""><a data-moreurl-dict='{"from":"top-nav-click-market","uid":"0"}' href="https://market.douban.com/?utm_campaign=douban_top_nav&amp;utm_source=douban&amp;utm_medium=pc_web" target="_blank">市集</a>
</li>
</ul>
'''
soup = BeautifulSoup(data,'lxml')
#
#print(soup.prettify())
# print(soup.title.name)
标签选择器 只返回匹配的第一个标签
# print(soup.head)
# print(soup.p['class'])
# print(soup.p.contents)
# print(soup.div.children)
# for i,child in enumerate(soup.ul.children):
#     print(i,child)
#获取父节点
# print(soup.a.parent)
#获取祖先节点
# print(soup.a.parents)
#兄弟节点
# print([i for i in enumerate(soup.li.next_siblings)])
标准选择器
# for li in soup.find_all('ul'):
#     print(li.find_all('li'))
#     print(len(li.find_all('li')))
#     print(type(li.find_all('li')))
CSS选择器
# print(soup.select('ul li'))
# print(soup.select('li a'))
# for i in soup.select('li a'):
#获取内容
#     print(i.get_text())
#     print(i.contents)
#获取属性
#     print(i.get('href'))
#     print(i.attrs['data-moreurl-dict'])
  • 0
    点赞
  • 0
    收藏
    觉得还不错? 一键收藏
  • 0
    评论
评论
添加红包

请填写红包祝福语或标题

红包个数最小为10个

红包金额最低5元

当前余额3.43前往充值 >
需支付:10.00
成就一亿技术人!
领取后你会自动成为博主和红包主的粉丝 规则
hope_wisdom
发出的红包
实付
使用余额支付
点击重新获取
扫码支付
钱包余额 0

抵扣说明:

1.余额是钱包充值的虚拟货币,按照1:1的比例进行支付金额的抵扣。
2.余额无法直接购买下载,可以购买VIP、付费专栏及课程。

余额充值