首页 > > 网络编程 > 其它 >

Python-爬虫小计

2019-01-21 02:42:55来源：博客园阅读 ()

# -*-coding:utf8-*-
import requests
from bs4 import BeautifulSoup
import time
import os
import urllib
import re
import json


requests.packages.urllib3.disable_warnings()

headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 6.1) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/59.0.3071.115 Safari/537.36'
}

def get_bs(url):
    res = requests.get(url, headers=headers, verify=False)
    bs = BeautifulSoup(res.content, 'lxml')
    return bs

def get_first_url():
    first_url_list = []
    page = 1
    for i in range(page):
        root_url =  "https://www.model61.com/mold.php?page={}".format(str(i+1))
        bs = get_bs(root_url)
        for i in  bs.select("dt a"):
            src = i.get('href')
            if "php" in src:
                first_url = "https://www.model61.com/{}".format(src)
                first_url_list.append(first_url)
    return first_url_list

def get_second_url(first_url):
    data = {}
    bs = get_bs(first_url)
    for i in bs.select(".cont-top a"):
        src = i.get('href')
        if "album_s" in src:
            second_url = "https://www.model61.com/{}".format(src)
            #print("second_url",second_url)
            data["second_url"] = second_url

    for j in bs.select(".content_center_date"):
        data["identity"] = j.get_text()
    return data


def get_thred_url(second_url):
    bs = get_bs(second_url)
    for i in  bs.select("dt a"):
        src = i.get('href')
        if "album_list" in src:
            thred_url = "https://www.model61.com/{}".format(src)
            #print("thred_url", thred_url)
            return thred_url


def get_image_list(thred_url):
    image_list = []
    bs = get_bs(thred_url)
    for i in bs.select(".album_list_left a")+bs.select(".album_list_right a"):
        src = i.get('href')
        image_path = "https://www.model61.com/{}".format(src)
        image_list.append(image_path)
        #print("image_path",image_path)
    return image_list

def download_image(image_path):
    pass

if __name__ == '__main__':
    first_url_list = get_first_url()
    for first_url in first_url_list:
        data = get_second_url(first_url)
        second_url = data['second_url']
        thred_url = get_thred_url(second_url)
        image_list = get_image_list(thred_url)
        data["image_list"] = image_list
        print(data)