python爬虫類はクールな犬の音楽ランキングを登ります

1222 ワード

#coding=utf-8
from pymongo import MongoClient
import time  
import requests  
from lxml import etree  

client = MongoClient()            #  mongo
hello = client.hello              #     
user = hello.song                 #   

headers = {  
    'User-Agent': 'Mozilla/5.0 (Android 6.0; Nexus 5 Build/MRA58N)\
    AppleWebKit/537.36 (KHTML, like Gecko) Chrome/65.0.3325.181 Mobile Safari/537.36'}  

def get_info(url):  
    '''
    get  ,encode,  ,xpath,   
    '''
    response = requests.get(url, headers=headers)  
    response = response.text.encode('utf-8')  
    selector = etree.HTML(response)  
    soup = selector.xpath('//*[@class="pc_temp_songlist "]/ul//li/a/text()') 

    #     
    # with open('aa.txt','a') as f:
        # for i in soup:
            # f.write(i.encode('utf-8') + '
') # for i in soup: user.insert({'song': i}) if __name__ == '__main__': urls = ['http://www.kugou.com/yy/rank/home/{}-8888.html?from=rank'.format(str(i)) for i in range(1, 24)] for url in urls: print(url) get_info(url)