#coding=utf-8
观察url,直接生成列表式爬取
import time import requests from lxml import etree headers = { 'User-Agent': 'Mozilla/5.0 (Android 6.0; Nexus 5 Build/MRA58N)\ AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Mobile Safari/537.36'} def get_info(url): ''' get源码,encode,解析,xpath,保存 ''' response = requests.get(url, headers=headers) response = response.text.encode('utf-8') selector = etree.HTML(response) soup = selector.xpath('//*[@class="pc_temp_songlist "]/ul//li/a/text()') with open('aa.txt','a') as f: for i in soup: f.write(i.encode('utf-8') + '\n') if __name__ == '__main__': urls = ['http://www.kugou.com/yy/rank/home/{}-8888.html?from=rank'.format(str(i)) for i in range(1, 24)] for url in urls: print(url) get_info(url)