Python爬去貼吧圖片

韓大帥666發表於2018-01-09

# tieba_xpath.py
#coding=utf-8

#!/usr/bin/env python
# -*- coding:utf-8 -*-

import os
import urllib
import urllib2
from lxml import etree

class Spider:
    def __init__(self):
        self.tiebaName = raw_input("請需要訪問的貼吧：")
        self.beginPage = int(raw_input("請輸入起始頁："))
        self.endPage = int(raw_input("請輸入終止頁："))

        self.url = 'http://tieba.baidu.com/f'
        self.ua_header = {"User-Agent" : "Mozilla/5.0 (compatible; MSIE 9.0; Windows NT 6.1 Trident/5.0;"}

        # 圖片編號
        self.userName = 1

    def tiebaSpider(self):
        for page in range(self.beginPage, self.endPage + 1):
            pn = (page - 1) * 50 # page number
            word = {'pn': pn, 'kw': self.tiebaName}

            word = urllib.urlencode(word) #轉換成url編碼格式（字串）
            myUrl = self.url + "?" + word

            # 示例：http://tieba.baidu.com/f? kw=%E7%BE%8E%E5%A5%B3 & pn=50
            # 呼叫 頁面處理函式 load_Page
            # 並且獲取頁面所有帖子連結,
            links = self.loadPage(myUrl)  # urllib2_test3.py

    # 讀取頁面內容
    def loadPage(self, url):
        req = urllib2.Request(url, headers = self.ua_header)
        html = urllib2.urlopen(req).read()

        # 解析html 為 HTML 文件
        selector=etree.HTML(html)

        #抓取當前頁面的所有帖子的url的後半部分，也就是帖子編號
        # http://tieba.baidu.com/p/4884069807裡的 “p/4884069807”
        links = selector.xpath('//div[@class="threadlist_lz clearfix"]/div/a/@href')

        # 驗證儲存的資料夾是否存在
        self.mkdir('C:/Users/Administrator/PycharmProjects/python/images/')
        # links 型別為 etreeElementString 列表
        # 遍歷列表，並且合併成一個帖子地址，呼叫 圖片處理函式 loadImage
        for link in links:
            link = "http://tieba.baidu.com" + link
            self.loadImages(link)

    # 獲取圖片
    def loadImages(self, link):
        req = urllib2.Request(link, headers = self.ua_header)
        html = urllib2.urlopen(req).read()

        selector = etree.HTML(html)

        # 獲取這個帖子裡所有圖片的src路徑
        imagesLinks = selector.xpath('//img[@class="BDE_Image"]/@src')

        # 依次取出圖片路徑，下載儲存
        for imagesLink in imagesLinks:
            self.writeImages(imagesLink)

    # 儲存頁面內容
    def writeImages(self, imagesLink):
        '''
            將 images 裡的二進位制內容存入到 userNname 檔案中
        '''

        print imagesLink
        print "正在儲存檔案 %d ..." % self.userName
        # 1. 開啟檔案，返回一個檔案物件
        file = open('C:/Users/Administrator/PycharmProjects/python/images/' + str(self.userName)  + '.png', 'wb')

        # 2. 獲取圖片裡的內容
        images = urllib2.urlopen(imagesLink).read()

        # 3. 呼叫檔案物件write() 方法，將page_html的內容寫入到檔案裡
        file.write(images)

        # 4. 最後關閉檔案
        file.close()

        # 計數器自增1
        self.userName += 1

    def mkdir(self,path):
        # 去除首位空格
        path = path.strip()
        # 去除尾部 \ 符號
        path = path.rstrip("/")

        # 判斷路徑是否存在
        # 存在     True
        # 不存在   False
        isExists = os.path.exists(path)

        # 判斷結果
        if not isExists:
            # 如果不存在則建立目錄
             # 建立目錄操作函式
            os.makedirs(path)

            print path + ' 建立成功'
            return True
        else:
            # 如果目錄存在則不建立，並提示目錄已存在
            print path + ' 目錄已存在'
            return False


# 模擬 main 函式
if __name__ == "__main__":

    # 首先建立爬蟲物件
    mySpider = Spider()
    # 呼叫爬蟲物件的方法，開始工作
    mySpider.tiebaSpider()

python爬去百度美女吧圖片
2018-04-01
Python
段友福利：Python爬取段友之家貼吧圖片和小視訊
2018-06-01
Python
lxml庫和貼吧圖片下載案例
2017-10-20
XML
網路爬蟲——爬百度貼吧
2015-12-28
爬蟲
Python爬蟲實戰（2）：百度貼吧帖子
2015-04-25
Python爬蟲
爬取百度貼吧實戰，python教你如何獲取
2020-12-07
Python
python爬蟲學習(2)-抓取百度貼吧內容
2017-02-15
Python爬蟲
Python爬蟲—爬取某網站圖片
2020-11-19
Python爬蟲網站
【python--爬蟲】千圖網高清背景圖片爬蟲
2019-05-21
Python爬蟲
Python爬蟲入門【5】：27270圖片爬取
2019-07-30
Python爬蟲
Python爬蟲學習（6）: 爬取MM圖片
2016-10-21
Python爬蟲
Python爬蟲之網頁圖片
2016-09-05
Python爬蟲網頁
Python 實用爬蟲-04-使用 BeautifulSoup 去水印下載 CSDN 部落格圖片
2019-06-16
Python爬蟲
Python爬蟲新手教程：知乎文章圖片爬取器
2019-07-20
Python爬蟲
Python爬蟲實戰詳解：爬取圖片之家
2020-11-04
Python爬蟲
Python爬蟲入門-爬取pexels高清圖片
2017-09-24
Python爬蟲
新手爬蟲教程：Python爬取知乎文章中的圖片
2019-01-17
爬蟲Python
Python爬蟲遞迴呼叫爬取動漫美女圖片
2020-10-19
Python爬蟲遞迴
Python 爬蟲入門 (二) 使用Requests來爬取圖片
2017-02-24
Python爬蟲
python 爬蟲之requests爬取頁面圖片的url，並將圖片下載到本地
2019-06-12
Python爬蟲
Python《必應bing桌面圖片爬取》
2020-12-26
Python
python3爬取1024圖片
2016-10-30
Python
Python爬蟲之煎蛋網圖片下載
2017-02-08
Python爬蟲
Python爬蟲搜尋並下載圖片
2017-12-13
Python爬蟲
小小圖片爬蟲
2016-12-01
爬蟲
Python 爬蟲零基礎教程(1)：爬單個圖片
2024-03-13
Python爬蟲
word貼上圖片到ckeitor
2021-04-16
Java爬蟲批量爬取圖片
2021-09-24
Java爬蟲
python爬蟲---網頁爬蟲，圖片爬蟲，文章爬蟲，Python爬蟲爬取新聞網站新聞
2019-01-04
Python爬蟲網頁網站
Python應用開發——爬取網頁圖片
2022-09-21
Python網頁
python爬蟲之圖片下載APP1.0
2016-12-22
Python爬蟲APP
Python爬取微博資料生成詞雲圖片
2017-08-29
Python
python 爬蟲下載百度美女圖片
2024-04-18
Python爬蟲
Python爬蟲入門【4】：美空網未登入圖片爬取
2019-07-30
Python爬蟲
Python爬蟲入門【6】：蜂鳥網圖片爬取之一
2019-07-30
Python爬蟲
Python爬蟲入門【7】：蜂鳥網圖片爬取之二
2019-07-31
Python爬蟲
Python爬蟲入門【8】：蜂鳥網圖片爬取之三
2019-07-31
Python爬蟲
Python網路爬蟲2 - 爬取新浪微博使用者圖片
2018-04-10
Python爬蟲

Python爬去貼吧圖片

相關文章