[轉載] python抓取電子書

時間 2019-11-13
原文原文鏈接
轉載自：https://blog.csdn.net/fei347795790/article/details/88541128html
from requests_html import HTMLSession
import requests
import time
import json
import random
import sys
import os

session = HTMLSession()
list_url = "http://www.allitebooks.com/page/"


USER_AGENTS = [
    "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_7_3) AppleWebKit/535.20 (KHTML, like Gecko) Chrome/19.0.1036.7 Safari/535.20",
    "Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.71 Safari/537.1 LBBROWSER",
    "Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/535.11 (KHTML, like Gecko) Chrome/17.0.963.84 Safari/535.11 LBBROWSER",
    "Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 6.1; WOW64; Trident/5.0; SLCC2; .NET CLR 2.0.50727; .NET CLR 3.5.30729; .NET CLR 3.0.30729; Media Center PC 6.0; .NET4.0C; .NET4.0E)",
    "Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E)",
    "Mozilla/4.0 (compatible; MSIE 7.0; Windows NT 5.1; Trident/4.0; SV1; QQDownload 732; .NET4.0C; .NET4.0E; 360SE)",
    "Mozilla/4.0 (compatible; MSIE 6.0; Windows NT 5.1; SV1; QQDownload 732; .NET4.0C; .NET4.0E)",
    "Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.1 (KHTML, like Gecko) Chrome/21.0.1180.89 Safari/537.1",
    "Mozilla/5.0 (iPad; U; CPU OS 4_2_1 like Mac OS X; zh-cn) AppleWebKit/533.17.9 (KHTML, like Gecko) Version/5.0.2 Mobile/8C148 Safari/6533.18.5",
    "Mozilla/5.0 (Windows NT 6.1; Win64; x64; rv:2.0b13pre) Gecko/20110307 Firefox/4.0b13pre",
    "Mozilla/5.0 (X11; Ubuntu; Linux x86_64; rv:16.0) Gecko/20100101 Firefox/16.0",
    "Mozilla/5.0 (Windows NT 6.1; WOW64) AppleWebKit/537.11 (KHTML, like Gecko) Chrome/23.0.1271.64 Safari/537.11",
    "Mozilla/5.0 (X11; U; Linux x86_64; zh-CN; rv:1.9.2.10) Gecko/20100922 Ubuntu/10.10 (maverick) Firefox/3.6.10"
]

def get_list(url):
    response = session.get(url)
    all_link = response.html.find('.entry-title a')
    for link in all_link:
        getBookUrl(link.attrs['href'])

def getBookUrl(url):
    response = session.get(url)
    l = response.html.find('.download-links a', first=True)
    f = getBookCategory(response)
    if l is not None:
        link = l.attrs['href']
        download(link, f)

def getBookCategory(response):
    res = response.html.find('.book-detail a')
    for item in res:
        if item.attrs['rel'][0] == 'category':
            return item.full_text


def download(url, folder):
    headers = {"User-Agent":random.choice(USER_AGENTS)}
    filename = url.split('/')[-1]
    if ".pdf" in url:
        if folder is not None:
            folder = 'book/' + folder
        else:
            folder = 'book/other'
        
        if os.path.exists(folder) is not True:
            os.mkdir(folder)
        file = folder + '/' + filename

        with open(file, 'wb') as f:
            print("Downloading %s" % filename)
            response = requests.get(url, stream=True, headers=headers)

            total_length = response.headers.get('content-length')
            if total_length is None:
                f.write(response.content)
            else:
                dl = 0
                total_length = int(total_length)
                for data in response.iter_content(chunk_size=4096):
                    dl += len(data)
                    f.write(data)
                    done = int(50 * dl / total_length)
                    sys.stdout.write("\r[%s%s]" % ('=' * done, ' ' * (50 - done)))
                    sys.stdout.flush()
            print(filename + "download complete!")

if __name__ == '__main__':
    for x in range(1, 100):
        print('Current page: {}/20'.format(str(x)))
        get_list(list_url + str(x))
相關標籤/搜索
每日一句
每一个你不满意的现在，都有一个你没有努力的曾经。