利用python选股票 (如何利用python分析股票)

功能描述:

  • 目标:获取上交所和深交所所有股票的名称和交易信息
  • 输出:保存到文件中
  • 技术路线:requests-bs4-re

用python做股票指标分析,利用python进行炒股

候选数据网站的选择:

  • 新浪股票:https://finance.sina.com.cn/stock/
  • 百度股票:https://gupiao.baidu.com/stock/
  • 选取原则:股票信息静态存在于HTML页面中,非js代码生成,没有Robots协议限制。

程序的结构设计

  • 步骤1:从东方财富网获取股票列表
  • 步骤2:根据股票列表逐个到百度股票获取个股信息
  • 步骤3:将结果存储到文件

用python做股票指标分析,利用python进行炒股

初步代码编写(error)

import requests
from bs4 import BeautifulSoup
import traceback
import re
'''
这是小编准备的python爬虫学习资料,加群:821460695 即可免费获取!
'''
def getHTMLText(url):
 try:
 r = requests.get(url)
 r.raise_for_status()
 r.encoding = r.apparent_encoding
 return r.text
 except:
 return ""
 
def getStockList(lst, stockURL):
 html = getHTMLText(stockURL)
 soup = BeautifulSoup(html, 'html.parser') 
 a = soup.find_all('a')
 for i in a:
 try:
 href = i.attrs['href']
 lst.append(re.findall(r"[s][hz]\d{6}", href)[0])
 except:
 continue
 
def getStockInfo(lst, stockURL, fpath):
 for stock in lst:
 url = stockURL + stock + ".html"
 html = getHTMLText(url)
 try:
 if html=="":
 continue
 infoDict = {}
 soup = BeautifulSoup(html, 'html.parser')
 stockInfo = soup.find('div',attrs={'class':'stock-bets'})
 
 name = stockInfo.find_all(attrs={'class':'bets-name'})[0]
 infoDict.update({'股票名称': name.text.split()[0]})
 
 keyList = stockInfo.find_all('dt')
 valueList = stockInfo.find_all('dd')
 for i in range(len(keyList)):
 key = keyList[i].text
 val = valueList[i].text
 infoDict[key] = val
 
 with open(fpath, 'a', encoding='utf-8') as f:
 f.write( str(infoDict) + '\n' )
 except:
 traceback.print_exc()
 continue
 
def main():
 stock_list_url = 'https://quote.eastmoney.com/stocklist.html'
 stock_info_url = 'https://gupiao.baidu.com/stock/'
 output_file = 'D:/BaiduStockInfo.txt'
 slist=[]
 getStockList(slist, stock_list_url)
 getStockInfo(slist, stock_info_url, output_file)
 
main()

代码优化(error)

速度提高:编码识别的优化

import requests
from bs4 import BeautifulSoup
import traceback
import re
'''
这是小编准备的python爬虫学习资料,加群:821460695 即可免费获取!
'''
def getHTMLText(url, code="utf-8"):
 try:
 r = requests.get(url)
 r.raise_for_status()
 r.encoding = code
 return r.text
 except:
 return ""
 
def getStockList(lst, stockURL):
 html = getHTMLText(stockURL, "GB2312")
 soup = BeautifulSoup(html, 'html.parser') 
 a = soup.find_all('a')
 for i in a:
 try:
 href = i.attrs['href']
 lst.append(re.findall(r"[s][hz]\d{6}", href)[0])
 except:
 continue
 
def getStockInfo(lst, stockURL, fpath):
 count = 0
 for stock in lst:
 url = stockURL + stock + ".html"
 html = getHTMLText(url)
 try:
 if html=="":
 continue
 infoDict = {}
 soup = BeautifulSoup(html, 'html.parser')
 stockInfo = soup.find('div',attrs={'class':'stock-bets'})
 
 name = stockInfo.find_all(attrs={'class':'bets-name'})[0]
 infoDict.update({'股票名称': name.text.split()[0]})
 
 keyList = stockInfo.find_all('dt')
 valueList = stockInfo.find_all('dd')
 for i in range(len(keyList)):
 key = keyList[i].text
 val = valueList[i].text
 infoDict[key] = val
 
 with open(fpath, 'a', encoding='utf-8') as f:
 f.write( str(infoDict) + '\n' )
 count = count + 1
 print("\r当前进度: {:.2f}%".format(count*100/len(lst)),end="")
 except:
 count = count + 1
 print("\r当前进度: {:.2f}%".format(count*100/len(lst)),end="")
 continue
 
def main():
 stock_list_url = 'https://quote.eastmoney.com/stocklist.html'
 stock_info_url = 'https://gupiao.baidu.com/stock/'
 output_file = 'D:/BaiduStockInfo.txt'
 slist=[]
 getStockList(slist, stock_list_url)
 getStockInfo(slist, stock_info_url, output_file)
 
main()

用python做股票指标分析,利用python进行炒股

测试成功代码

由于东方财富网链接访问时出现错误,所以更换了一个新的网站去获取股票列表,具体代码如下:

import requests
import re
import traceback
from bs4 import BeautifulSoup
import bs4
def getHTMLText(url):
 try:
 r = requests.get(url, timeout=30)
 r.raise_for_status()
 r.encoding = r.apparent_encoding
 return r.text
 except:
 return""
def getStockList(lst, stockListURL):
 html = getHTMLText(stockListURL)
 soup = BeautifulSoup(html, 'html.parser')
 a = soup.find_all('a')
 lst = []
 for i in a:
 try:
 href = i.attrs['href']
 lst.append(re.findall(r"[S][HZ]\d{6}", href)[0])
 except:
 continue
 lst = [item.lower() for item in lst] # 将爬取信息转换小写
 return lst
def getStockInfo(lst, stockInfoURL, fpath):
 count = 0
 for stock in lst:
 url = stockInfoURL + stock + ".html"
 html = getHTMLText(url)
 try:
 if html == "":
 continue
 infoDict = {}
 soup = BeautifulSoup(html, 'html.parser')
 stockInfo = soup.find('div', attrs={'class': 'stock-bets'})
 if isinstance(stockInfo, bs4.element.Tag): # 判断类型
 name = stockInfo.find_all(attrs={'class': 'bets-name'})[0]
 infoDict.update({'股票名称': name.text.split('\n')[1].replace(' ','')})
 keylist = stockInfo.find_all('dt')
 valuelist = stockInfo.find_all('dd')
 for i in range(len(keylist)):
 key = keylist[i].text
 val = valuelist[i].text
 infoDict[key] = val
 with open(fpath, 'a', encoding='utf-8') as f:
 f.write(str(infoDict) + '\n')
 count = count + 1
 print("\r当前速度:{:.2f}%".format(count*100/len(lst)), end="")
 except:
 count = count + 1
 print("\r当前速度:{:.2f}%".format(count*100/len(lst)), end="")
 traceback.print_exc()
 continue
def main():
 fpath = 'D://gupiao.txt'
 stock_list_url = 'https://hq.gucheng.com/gpdmylb.html'
 stock_info_url = 'https://gupiao.baidu.com/stock/'
 slist = []
 list = getStockList(slist, stock_list_url)
 getStockInfo(list, stock_info_url, fpath)
main()