1.爬虫的流程?
1.爬虫的流程?
第一步,获取网页内容 http请求,request发送请求 第二步,解析网页内容 第三步,储存或分析数据
2.什么是HTTP请求和响应?
Get方法 ==请求行== POST /user/info HTTP/1.1 请求行
==请求头==
Host: www.example.com
User-Agent: curl/7.77.0
Accept:
/
==请求体==
“username":"fei""email": "
l***@qq.com"
http响应
3.如何用Python Requests发送请求?
先安装requests
pip install requests
import requests
response = requests.get("http://books.toscrape.com")
print(response)
print(response.status_code)
if response.status_code >= 200 and response.status_code < 400:
# 获取响应体内容
elif response.status_code >= 400 and response.status_code < 500:
print("请求失败,客户端错误")
elif response.status_code >= 500:
print("请求失败,服务器错误")
简化
if response.ok:
print(response.text)
else:
print("请求失败")
功能代码
import requests
head = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
response = requests.get("http://books.toscrape.com",headers=head)
if response.ok:
print(response.text)
else:
print("请求失败")
4.实践: 用Python Requests拿到豆瓣源码
如果没有请求头,很可能回应一个418的错误状态码 请求头可以在浏览器网络工具-网络里找到,任一个请求点开,找到User-Agent。本机Edge浏览器请求头如下: Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/117.0.x.x Safari/537.36 Edg/117.0.2045.36
5.如何用Beautiful Soup解析内容?
Beautiful Soup是第三方库,需要安装
pip install bs4
Beautiful Soup可以将看起来复杂的html内容解析成树状结构,然后获取对应标签的所有内容
下面的代码是获取网页中的书籍价格
from bs4 import BeautifulSoup
import requests
head = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
content = requests.get("http://books.toscrape.com").text
soup = BeautifulSoup(content,"html.parser")
all_prices = soup.find_all("p",attrs={"class":"price_color"})
# findall 可返回一个可迭代对象,在这里返回的是p标签中属性class为price_color的内容
for price in all_prices:
# print(price)
print(price.string[2:]) # string属性会把标签包围的文字返回,还能进行切片
获取网页中书籍名称
from bs4 import BeautifulSoup
import requests
head = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
content = requests.get("http://books.toscrape.com").text
soup = BeautifulSoup(content,"html.parser")
all_titles = soup.find_all("h3")
# findall 返回书名对应的h3标签内容
for title in all_titles:
\#all_links = title.findAll("a")
# for link in all_links:
# print(link.string)
# 因为在这个网页中,只有一个a元素,也可以使用find
link = title.find("a")
print(link.string)
6.实践: 从源码获取豆瓣电影Top 250
获取其中一页电影标题的代码如下:
from bs4 import BeautifulSoup
import requests
head = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
content = requests.get("https://movie.douban.com/top250",headers=head).text
soup = BeautifulSoup(content,"html.parser")
all_titles = soup.find_all("span",attrs={"class":"title"})
# findall 返回豆瓣中对应hd的电影标题
# print(all_titles)
for title in all_titles:
if "/" not in title.string: # 剔除外文标题
print(title.string)
但是目前只有获取第一页的内容,通过分析得知 如果要获取前250个信息,修改代码如下:
from bs4 import BeautifulSoup
import requests
head = {"User-Agent":"Mozilla/5.0 (Windows NT 10.0; Win64; x64)"}
f = open('test.txt','a') # 保存至当前文件夹
for start_num in range(0,250,25): # 获取0~250编号的电影,每页25页信息
# print(start_num)
content = requests.get(f"https://movie.douban.com/top250?start={start_num}",headers=head).text
# 更新链接信息
soup = BeautifulSoup(content,"html.parser")
all_titles = soup.find_all("span",attrs={"class":"title"})
# findall 返回豆瓣中对应hd的电影标题
# print(all_titles)
for title in all_titles:
if "/" not in title.string: # 剔除外文标题
f.writelines(title.string)
7.拓展代码
github上搜索到的代码,需要提前用pip install requests BeautifulSoup xlwt
import requests
from bs4 import BeautifulSoup
import xlwt
def request_douban(url):
headers = {
'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) '
'Chrome/88.0.4324.146 Safari/537.36',
}
try:
response = requests.get(url=url, headers=headers)
if response.status_code == 200:
return response.text
except requests.RequestException:
return None
book = xlwt.Workbook(encoding='utf-8', style_compression=0)
sheet = book.add_sheet('豆瓣电影Top250', cell_overwrite_ok=True)
sheet.write(0, 0, '名称')
sheet.write(0, 1, '图片')
sheet.write(0, 2, '排名')
sheet.write(0, 3, '评分')
sheet.write(0, 4, '作者')
sheet.write(0, 5, '简介')
n = 1
def save_to_excel(soup):
list = soup.find(class_='grid_view').find_all('li')
for item in list:
item_name = item.find(class_='title').string
item_img = item.find('a').find('img').get('src')
item_index = item.find(class_='').string
item_score = item.find(class_='rating_num').string
item_author = item.find('p').text
if item.find(class_='inq') is not None:
item_intr = item.find(class_='inq').string
else:
item_intr = 'NOT AVAILABLE'
# print('爬取电影:' + item_index + ' | ' + item_name +' | ' + item_img +' | ' + item_score +' | ' + item_author +' | ' + item_intr )
print('爬取电影:' + item_index + ' | ' + item_name + ' | ' + item_score + ' | ' + item_intr)
global n
sheet.write(n, 0, item_name)
sheet.write(n, 1, item_img)
sheet.write(n, 2, item_index)
sheet.write(n, 3, item_score)
sheet.write(n, 4, item_author)
sheet.write(n, 5, item_intr)
n = n + 1
def main(page):
url = 'https://movie.douban.com/top250?start=' + str(page * 25) + '&filter='
html = request_douban(url)
soup = BeautifulSoup(html, 'lxml')
save_to_excel(soup)
if __name__ == '__main__':
for i in range(0, 10):
main(i)
book.save(u'豆瓣最受欢迎的250部电影.xlsx')