爬虫入门二(续一)
文末附教程博客链接,感兴趣可以去看一下。
用html文件保存爬取到的数据
python代码:
import requests
from bs4 import BeautifulSoup
#1-1.获取网页信息保存到文件的过程
#url = "https://movie.douban.com/cinema/later/chengdu/"
#response = requests.get(url)
#file_obj = open('douban.html','w',encoding="utf-8")
#file_obj.write(response.content.decode('utf-8'))
#file_obj.close()
#1-2.从文件获取信息的过程
#file_obj = open('douban.html','r', encoding="utf-8")
#html = file_obj.read()
#file_obj.close()
#1-3.初始化BeautifulSoup,解析网页
#soup = BeautifulSoup(html, 'lxml')
#print(soup.find)
#2.直接抓取、解析
url = "https://movie.douban.com/cinema/later/chengdu/"
response = requests.get(url)
soup = BeautifulSoup(response.content.decode('utf-8'), 'lxml')
#3.获取并分析元素
all_movies = soup.find(<