- https://www.1ppt.com/moban/
# ● 爬取要求:
# ○ 1、 翻页爬取这个网页上面的源代码
# ○ 2、 并且保存到本地,注意编码">作业1
# ● 目标网站:https://www.1ppt.com/moban/
# ● 爬取要求:
# ○ 1、 翻页爬取这个网页上面的源代码
# ○ 2、 并且保存到本地,注意编码
作业1
# ● 目标网站:https://www.1ppt.com/moban/
# ● 爬取要求:
# ○ 1、 翻页爬取这个网页上面的源代码
# ○ 2、 并且保存到本地,注意编码
import urllib.request
import ssl
# url = ‘https://www.1ppt.com/moban/‘
headers = {
‘User-Agent’: ‘Mozilla/5.0 (Macintosh; Intel Mac OS X 1015_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/99.0.4844.74 Safari/537.36’
}
start = int(input(‘请输入你要开始页码:’))
end = int(input(‘请输入你要结束页码:’))
for i in range(start, end + 1):
url = f’https://www.1ppt.com/moban/ppt_moban{i}.html’
# 因为是https协议,所以要关闭协议才可以
ssl._create_default_https_context = ssl._create_unverified_context
result = urllib.request.Request(url, headers=headers)
response = urllib.request.urlopen(result)
res_r = response.read()
# 保存数据(保存本地)
with open(f’第{i}页的.html’, ‘w’, encoding=’gb2312’) as f:
f.write(res_r.decode(‘gb2312’))
�
