Python爬蟲爬取百度貼吧的帖子

一樣是參考網上教程,編寫爬取貼吧帖子的內容,同時把爬取的帖子保存到本地文檔:python

 

#!/usr/bin/python
#_*_coding:utf-8_*_
import urllib
import urllib2
import re
import sysapp

reload(sys)
sys.setdefaultencoding("utf-8")
#處理頁面標籤,去除圖片、超連接、換行符等
class Tool:
#去除img標籤,7位長空格
removeImg = re.compile('<img.*?>| {7}|')
#刪除超連接標籤
removeAddr = re.compile('<a.*?>|</a>')
#把換行的標籤替換爲\n
replaceLine = re.compile('<tr>|<div>|</div>|</p>')
#把表格製表<td>替換爲\t
replaceTD = re.compile('<td>')
#把段落開頭換爲\n加兩個空格
replacePara = re.compile('<p.*?>')
#把換行符或雙換行符替換爲\n
replaceBR = re.compile('<br><br>|<br>')
#將其他標籤剔除
removeET = re.compile('<.*?>')工具

#去除匹配到Tool
def replace(self,x):
x = re.sub(self.removeImg,"",x)
x = re.sub(self.removeAddr,"",x)
#x = re.sub(self.replaceLine,"\n",x)
#x = re.sub(self.replaceTD,"\t",x)
#x = re.sub(self.replacePara,"\n ",x)
#x = re.sub(self.replaceBR,"\n",x)
x = re.sub(self.removeET,"",x)
#strip()將先後多餘內容刪除
return x.strip().encode('utf-8')
#百度貼吧爬蟲練習
class BDTB:post

#初始化,傳入地址,是否只看樓主的參數
def __init__(self,baseUrl,seeLz,floorTag):
#base連接地址
self.baseURL = baseUrl
#是否只看樓主
self.seeLZ = '?seelz=' + str(seeLz)
#HTML剔除標籤工具Tool
self.tool = Tool()
#全局file變量,文件寫入操做對象
self.file = None
#樓層標識,初始化爲1
self.floor = 1
#默認的標題,若是沒有成功獲取到標題的話則會用這個標題
self.defaultTitle = u"百度貼吧"
#是否寫入樓分隔符的標記
self.floorTag = floorTagurl

#傳入頁碼,獲取該頁帖子的代碼
def getPage(self,pageNum):
try:
url = self.baseURL + self.seeLZ + '&pn=' + str(pageNum)
request = urllib2.Request(url)
response = urllib2.urlopen(request)
tbPage = response.read().decode('utf-8')
#print tbPage
return tbPage
#連接報錯的緣由
except urllib2.URLError, e:
if hasattr(e,"reason"):
print u'連接百度貼吧失敗,錯誤緣由:',e.reason
return Nonespa

#獲取帖子標題
def getTitle(self,page):
page = self.getPage(1)
#正則匹配貼吧標題
pattern = re.compile('<h3 class="core_title_txt.*?>(.*?)</h3>',re.S)
result = re.search(pattern,page)
if result:
#輸出標題
#print result.group(1)
return result.group(1).strip()
else:
return Nonecode

#獲取帖子一共有多少頁
def getPageNum(self,page):
page = self.getPage(1)
#正則匹配帖子總共有多少頁
pattern = re.compile('<li class="l_reply_num.*?</span>.*?<span.*?>(.*?)</span>',re.S)
result = re.search(pattern,page)
if result:
#輸出頁碼數
#print result.group(1)
return result.group(1).strip()
else:
print None對象

#獲取帖子每個樓層的內容
def getContent(self,page):
#正則匹配每個樓層的內容
pattern = re.compile('<div id="post_content.*?>(.*?)</div>',re.S)
items = re.findall(pattern,page)
#floor = 1
contents = []
for item in items:
#將文本進行去除標籤處理,同時在先後加入換行符
content = "\n" + self.tool.replace(item) + "\n"
contents.append(content.encode('utf-8'))
#print floor,u"樓-----------------------"
#print content
#floor += 1
return contents教程

#設置文件的標題
def setFileTitle(self,title):
#若是標題不是None,即成功獲取到標題
if title is not None:
self.file = open(title + ".txt","w+")
else:
self.file = open(self.defaultTitle + ".txt","w+")圖片

#向文件寫入每一樓層的信息
def writeData(self,contents):
#遍歷樓層
for item in contents:
if self.floorTag == '1':
#樓之間使用的分隔符
floorLine = "\n--------------" + str(self.floor) + "樓-----------------\n"
self.file.write(unicode(floorLine,"utf-8"))
self.file.write(unicode(item,"utf-8"))
self.floor += 1

def start(self): indexPage = self.getPage(1) pageNum = self.getPageNum(indexPage) title = self.getTitle(indexPage) self.setFileTitle(title) if pageNum == None: print "URL已失效,請重試" return try: print "該帖子共有" + str(pageNum) + "頁" for i in range(1,int(pageNum) + 1): print "正在寫入第" + str(i) + "頁數據" page = self.getPage(i) contents = self.getContent(page) self.writeData(contents) except IOError,e: print "寫入異常,緣由" + e.message finally: print "寫入任務完成"print u"請輸入帖子代號"baseURL = 'http://tieba.baidu.com/p/' + str(raw_input(u'http://tieba.baidu.com/p/'))seeLZ = raw_input("是否只獲取樓主發言,是輸入1,否輸入0\n")floorTag = raw_input("是否寫入樓層信息,是輸入1,否輸入0\n")bdtb = BDTB(baseURL,seeLZ,floorTag)bdtb.start()

相關文章
相關標籤/搜索