1. # 需求:爬取58二手房中的房源信息
    2. if __name__ == "__main__":
    3. headers = {
    4. 'User-Agent': 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_12_0) AppleWebKit/537.36 (KHTML, like Gecko) '
    5. 'Chrome/73.0.3683.103 Safari/537.36 '
    6. }
    7. # 爬取到页面源码数据
    8. url = 'https://bj.58.com/ershoufang/'
    9. page_text = requests.get(url=url, headers=headers).text
    10. # 数据解析
    11. tree = etree.HTML(page_text)
    12. # 存储的就是li标签对象
    13. li_list = tree.xpath('//ul[@class="house-list-wrap"]/li')
    14. fp = open('58.txt', 'w', encoding='utf-8')
    15. for li in li_list:
    16. # 局部解析
    17. title = li.xpath('./div[2]/h2/a/text()')[0]
    18. print(title)
    19. fp.write(title + '\n')