Tag:fresh pre usr Get des oca soup load lis
#!/usr/bin/python#-*-Coding:utf8-*-import requestsimport reimport os import time # from Urllibimport jsonfrom BS4 impo RT beautifulsoupfrom DateTime Import datedef gettimeexpire (time_play,time_gap): # print (Time_play) try:time_arr= Time.strptime (Time_play, "%y-%m-%d%h:%m:%s") except:print (' Time conversion failed ') return ' Else:t1=time.mktime (time_arr) x = Time.localtime (T1+TIME_GAP) #是秒不是毫秒return time.strftime ('%y-%m-%d%h:%m:%s ', x) def gethtml (): #改成从网站直接获取, But the site needs to be paged with open (' f:\\test\\python\\worldcup.html ', ' R ', encoding= ' Utf-8 ') as F:content = F.read () soup = BeautifulSoup (Content, ' lxml ') nodes=soup.select ('. b-pull-refresh-content > div ') arr=[] #写入CSV文件的头部filename = "F:\\test\\python \\worldcup.csv "f = open (filename, ' a ') f.writelines (' Team1,team2,time_expire,time_play \ n ') f.close () for node in nodes: Date = Node.select ('. Wa-match-schedule-list-title ') [0].get_text (). Strip () Datas = Node.select ('. Sfc-contacts-list. Wa-match-schedule-list-item ') for D in datas:obj={' team1 ': ', ' team2 ': ', ' time ': '}obj[' team1 ']=d.Select ('. Wa-tiyu-schedule-item-name.c-line-clamp1 ') [0].get_text (). Strip () obj[' team2 ']=d.select ('. Wa-tiyu-schedule-item-name.c-line-clamp1 ') [1].get_text (). Strip () obj[' time_play ']= ' 2018-' +date[2:8]+ ' +d.select ('. Status-text ') [0].get_text (). Strip () + ': ' obj[' Time_expire ']=gettimeexpire (obj[' time_play '],-10*60) filename = "f:\\test\\ Python\\worldcup.csv "f = open (filename, ' a ') f.writelines (obj[' team1 ']+ ', ' +obj[' team2 ']+ ', ' +obj[' time_expire ']+ ', ' +obj[' time_play ']+ ' \ n ') f.close () #getHtml () def getfromapi (): month=6day=11# from 2018-06-14 to 07-15for D in range (0,15): Day +=2if day>30:month+=1day=1url= "http://tiyu.baidu.com/api/match/%e4%b8%96%e7%95%8c%e6%9d%af/live/date/2018-" + STR (month) + '-' +str (day) + "/direction/after?from=self" Time.sleep (1) data = Json.loads (Requests.get (url,timeout=3). Text) if (data[' status ']== ' 0 '):p rint (' 0 ') for matches in data[' data ']:for m in matches[' list ']:filename = "f:\\test\\ Python\\worldcupfromapi.csv "f = open (filename, ' a ') if m[' StartTime ']>time.strftime ("%y-%m-%d%h:%m:%s", Time.localtime ()): F.writelines (m[' Leftlogo ' [' Name ']+ ', ' +m[' Rightlogo '] ' name ']+ ', ' +gettimeexpire (m[') StartTime '],-10*60) + ', ' +m[' startTime ']+ ' \ n ') f.close () Getfromapi ()
Python crawler gets World Cup match schedule