Python 爬蟲web網頁版程式碼

來源:互聯網
上載者:User

一:網頁結構分析

二:代碼實戰

#! /usr/bin/env python2# encoding=utf-8#BeautifulSoup需要安裝 MySQLdbimport sys,os,re,hashlibimport urllibimport httplib2from lxml import etreeimport MySQLdbfrom BeautifulSoup import BeautifulSoupimport urllib2import reimport timereload(sys)from datetime import datetime as dt,timedeltaimport re h=httplib2.Http(timeout=10)#佈建要求http頭 類比偽裝 瀏覽器headers={    'User-Agent':'Mozilla/4.0 (compatible; MSIE 8.0; Windows NT 6.0; Trident/4.0)'}#正則匹配a標籤pattern = '<a.*?href="(.+)".*?>(.*?)</a>'#日誌記錄log_path='./sporttery'log_file='%s.log' % dt.now().strftime('%Y-%m-%d')if not os.path.exists(log_path):    os.makedirs(log_path)log=open('%s/%s' % (log_path,log_file),'w+') #python操作mysql資料庫conn= MySQLdb.connect(        host='localhost',        port = 3306,        user='root',        passwd='root',        db ='test',        )conn.set_character_set('utf8')cur = conn.cursor()cur.execute('SET NAMES utf8;')cur.execute('SET CHARACTER SET utf8;')cur.execute('SET character_set_connection=utf8;')cur.close() #擷取請求連結內容 失敗再次執行def download(url):    fails = 0    while True:        if fails>5:return None        try:            res,content = h.request(url,'GET',headers=headers)            return  content.decode('utf-8','ignore')        except:            print(u'開啟連結失敗'+url)            fails +=1#字串截取方法 def GetMiddleStr(content,startStr,endStr):  startIndex = content.index(startStr)  if startIndex>=0:    startIndex += len(startStr)  endIndex = content.index(endStr)  return content[startIndex:endIndex]def get_ul(data):    mystring=GetMiddleStr(data,'<ul class="cenicon">','<div class="clear hed"></div>')    return mystring def test_sporttery(i):    url='http://www.xxx.com/video/video_%E8%B6%B3%E7%90%83%E9%AD%94%E6%96%B9_'+str(i)+'.html'    print url    #http://www.xxx.com/video/video_%E8%B6%B3%E7%90%83%E9%AD%94%E6%96%B9_2.html    source=download(url)    data=get_ul(source)    datas=data.split('<li>')    for each in datas:        ret=re.findall(r"(?<=href=\").+?(?=\")|(?<=href=\').+?(?=\')" ,each)        for urls in ret:            detial=download(urls)            if detial:                detial_content=GetMiddleStr(detial,'createFlashVideo','m3u8').replace(' ', '')                if detial_content:                    end_url_rex=GetMiddleStr(detial_content+".m3u8",'http://','.m3u8')+"m3u8"                    #最終的url                    #title                    sstree = etree.HTML(detial)                    ssnodes = sstree.xpath('//*[@id="playVideo"]/div[1]/h2')                    for ssn in ssnodes:                        name= ssn.text.strip().replace('/h2>', '')                    #title=GetMiddleStr(detial,'<h3','</h3>').replace(' ', '')                    #簡介                    introduction=GetMiddleStr(detial,'video-info">','<!-- /.w1000 -->').replace(' ', '')                    dr = re.compile(r'<[^>]+>',re.S)                    introductions = dr.sub('',introduction)                    end_content=introductions.strip().replace('/span>', '')                    end_time= time.strftime("%Y-%m-%d %H:%M:%S", time.localtime(time.time()+8*60*60))                    #end_times=dt.now().strftime('%Y-%m-%d %H:%i:%S')                    saveDB(urls,end_url_rex,name,end_content,str(i),end_time)  def saveDB(current_url,end_url_rex,names,end_content,page,create_time):    #添加select update    sql = 'INSERT INTO test.mytables(current_url,end_url_rex,`names`,end_content,page,create_time)\            VALUES (%s,%s,%s,%s,%s,%s)'    print sql    cur = conn.cursor()    cur.execute(sql,(current_url,end_url_rex,names,end_content,page,create_time))    cur.close()    conn.commit() if __name__ == '__main__':     first="http://www.xxx.com/video/video_%E8%B6%B3%E7%90%83%E9%AD%94%E6%96%B9_1.html"     url = urllib2.urlopen(first)     content = url.read()     soup = BeautifulSoup(content)     strs=soup.findAll(attrs={"class":"pagination"})     lists=str(strs[0])     listss=re.findall(r'\d+',lists)     count=len(listss)     list_string = list(set(listss))     str_num= list_string[-1]     i = 1     while i <= int(str_num):           test_sporttery(i)           i += 1

聯繫我們

該頁面正文內容均來源於網絡整理,並不代表阿里雲官方的觀點,該頁面所提到的產品和服務也與阿里云無關,如果該頁面內容對您造成了困擾,歡迎寫郵件給我們,收到郵件我們將在5個工作日內處理。

如果您發現本社區中有涉嫌抄襲的內容,歡迎發送郵件至: info-contact@alibabacloud.com 進行舉報並提供相關證據,工作人員會在 5 個工作天內聯絡您,一經查實,本站將立刻刪除涉嫌侵權內容。

A Free Trial That Lets You Build Big!

Start building with 50+ products and up to 12 months usage for Elastic Compute Service

  • Sales Support

    1 on 1 presale consultation

  • After-Sales Support

    24/7 Technical Support 6 Free Tickets per Quarter Faster Response

  • Alibaba Cloud offers highly flexible support services tailored to meet your exact needs.