MOOC 中国大学 python爬取股票信息

xiaoxiao2021-02-28  75

#该程序先从东方财富网爬取股票列表 然后在从百度股票的找出对应股票的信息 import requests from bs4 import BeautifulSoup import traceback import re def getHTMLText(url, code="utf-8"): try: r = requests.get(url) r.raise_for_status() r.encoding = code#直接告诉编码类型 加快程序速度 return r.text except: return "" def getStockList(lst, stockURL): html = getHTMLText(stockURL, "GB2312")#告诉编码类型 加快程序速度 soup = BeautifulSoup(html, 'html.parser') a = soup.find_all('a')#找出所有a标签 for i in a: try: href = i.attrs['href']#把a标签中所有href 存到数组href中 lst.append(re.findall(r"[s][hz]\d{6}", href)[0])#用正则表达式提取href中股票编号 存到lst中 except: continue def getStockInfo(lst, stockURL, fpath): count = 0#计数 for stock in lst: url = stockURL + stock + ".html"#每一只股票的html页面 html = getHTMLText(url) try: if html=="":#页面为空 继续爬取其他 continue infoDict = {} soup = BeautifulSoup(html, 'html.parser')#解析html stockInfo = soup.find('div',attrs={'class':'stock-bets'})#找出div 标签中属性为stock-bets的 name = stockInfo.find_all(attrs={'class':'bets-name'})[0]#找出所有属性为 bets-name 并存到name列表中 infoDict.update({'股票名称': name.text.split()[0]})#lst 填入 name中空格前的内容(真正的name) keyList = stockInfo.find_all('dt')#找出与股票相关的键值对列表 valueList = stockInfo.find_all('dd') for i in range(len(keyList)): key = keyList[i].text#还原成键 val = valueList[i].text#还原值 infoDict[key] = val #向字典中新增内容 with open(fpath, 'a', encoding='utf-8') as f:#保存到文件中 f.write( str(infoDict) + '\n' ) count = count + 1 print("\r当前进度: {:.2f}%".format(count*100/len(lst)),end="")#进度条 except: count = count + 1 print("\r当前进度: {:.2f}%".format(count*100/len(lst)),end="") traceback.print_exc()#获得错误信息 continue def main(): stock_list_url = 'http://quote.eastmoney.com/stocklist.html' stock_info_url = 'https://gupiao.baidu.com/stock/' output_file = 'D:/BaiduStockInfo1.txt' slist=[] getStockList(slist, stock_list_url) getStockInfo(slist, stock_info_url, output_file) main()
转载请注明原文地址: https://www.6miu.com/read-97264.html

最新回复(0)