import requests
from bs4 import BeautifulSoup
import bs4
def getHTMLText(url):
try:
r = requests.get(url, timeout=30)
r.raise_for_status()
r.encoding = r.apparent_encoding
return r.text
except:
return "通路出錯"
def fillUnivList(ulist,html):
soup = BeautifulSoup(html, "html.parser")
for tr in soup.find('tbody').children:
if isinstance(tr, bs4.element.Tag):
tds = tr('td')
ulist.append([tds[0].string,tds[1].string,tds[2].string,tds[3].string,tds[4].string])
def printUnivList(ulist,num):
tplt="{0:^10}\t{1:{5}^10}\t{2:^10}\t{3:^10}\t{4:^10}" #中英文對齊,格式化
print(tplt.format("排名","學校名稱","省份","總分","名額得分",chr(12288)))
for i in range(num):
u = ulist[i]
print(tplt.format(u[0],u[1],u[2],u[3],u[4],chr(12288)))
def main():
uinfo = []
url = 'http://www.zuihaodaxue.cn/zuihaodaxuepaiming2016.html'
html = getHTMLText(url)
print(html)
fillUnivList(uinfo,html)
printUnivList(uinfo, 20)
main()