## This script uses a quandl account to scrape  historical stock data 
### I chose to look at stocks in the S&P 500 and arbitrarily chose a date in 2015 to determine which S&P companies

Going through the S&P 500 took 3.5 hours to scrape

In [12]:
import pandas as pd
import numpy as np
import quandl
import datetime
import pword
import time
from tqdm import tqdm

In [4]:
quandl.ApiConfig.api_key = pword.password

In [1]:
def quandl_stocks(symbol, start_date=(2000, 1, 1)):
    """
   Symbol is a string representing a stock symbol, e.g. 'AAPL' start_date and end_date are tuples of integers representing the year, month,
   and day end_date defaults to the current date when None
    """
    #The numbers in this query list are for the columns of interest
    query_list = ['WIKI' + '/' + symbol + '.' + str(k) for k in range(1, 8)]
    start_date = datetime.date(*start_date)
    end_date = datetime.date.today()
    return quandl.get(query_list, 
            returns='pandas', 
            start_date=start_date,
            end_date=end_date,
            collapse='daily',
            order='asc'
            )

In [26]:
#496 of the s&p 500 stocks (as of 7/7/2015)
sp500 = ['AAPL', 'ABT', 'ABBV', 'ACN', 'ACE', 'ADBE', 'ADT', 'AAP', 'AES', 'AET', 'AFL',
         'AMG', 'A', 'GAS', 'ARE', 'APD', 'AKAM', 'AA', 'AGN', 'ALXN', 'ALLE', 'ADS', 'ALL', 
         'ALTR', 'MO', 'AMZN', 'AEE', 'AAL', 'AEP', 'AXP', 'AIG', 'AMT', 'AMP', 'ABC', 'AME', 'AMGN', 'APH', 
         'APC', 'ADI', 'AON', 'APA', 'AIV', 'AMAT', 'ADM', 'AIZ', 'T', 'ADSK', 'ADP', 'AN', 'AZO', 'AVGO', 'AVB', 
         'AVY', 'BHI', 'BLL', 'BAC', 'BK', 'BCR', 'BXLT', 'BAX', 'BBT', 'BDX', 'BBBY', 'BRK.B', 'BBY', 'BLX', 'HRB', 'BA', 'BWA', 'BXP', 'BSX', 'BMY', 'BRCM', 'BF.B', 'CHRW', 'CA', 'CVC', 'COG', 'CAM', 'CPB', 'COF', 'CAH', 'HSIC', 'KMX', 'CCL', 'CAT', 'CBG', 'CBS', 'CELG', 'CNP', 'CTL', 'CERN', 'CF', 'SCHW', 'CHK', 'CVX', 'CMG', 'CB', 'CI', 'XEC', 'CINF', 'CTAS', 'CSCO', 'C', 'CTXS', 'CLX', 'CME', 'CMS', 'COH', 'KO', 'CCE', 'CTSH', 'CL', 'CMCSA', 'CMA', 'CSC', 'CAG', 'COP', 'CNX', 'ED', 'STZ', 'GLW', 'COST', 'CCI', 'CSX', 'CMI', 'CVS', 'DHI', 'DHR', 'DRI', 'DVA', 'DE', 'DLPH', 'DAL', 'XRAY', 'DVN', 'DO', 'DTV', 'DFS', 'DISCA', 'DISCK', 'DG', 'DLTR', 'D', 'DOV', 'DOW', 'DPS', 'DTE', 'DD', 'DUK', 'DNB', 'ETFC', 'EMN', 'ETN', 'EBAY', 'ECL', 'EIX', 'EW', 'EA', 'EMC', 'EMR', 'ENDP', 'ESV', 'ETR', 'EOG', 'EQT', 'EFX', 'EQIX', 'EQR', 'ESS', 'EL', 'ES', 'EXC', 'EXPE', 'EXPD', 'ESRX', 'XOM', 'FFIV', 'FB', 'FAST', 'FDX', 'FIS', 'FITB', 'FSLR', 'FE', 'FISV', 'FLIR', 'FLS', 'FLR', 'FMC', 'FTI', 'F', 'FOSL', 'BEN', 'FCX', 'FTR', 'GME', 'GPS', 'GRMN', 'GD', 'GE', 'GGP', 'GIS', 'GM', 'GPC', 'GNW', 'GILD', 'GS', 'GT', 'GOOGL', 'GOOG', 'GWW', 'HAL', 'HBI', 'HOG', 'HAR', 'HRS', 'HIG', 'HAS', 'HCA', 'HCP', 'HCN', 'HP', 'HES', 'HPQ', 'HD', 'HON', 'HRL', 'HSP', 'HST', 'HCBK', 'HUM', 'HBAN', 'ITW', 'IR', 'INTC', 'ICE', 'IBM', 'IP', 'IPG', 'IFF', 'INTU', 'ISRG', 'IVZ', 'IRM', 'JEC', 'JBHT', 'JNJ', 'JCI', 'JOY', 'JPM', 'JNPR', 'KSU', 'K', 'KEY', 'GMCR', 'KMB', 'KIM', 'KMI', 'KLAC', 'KSS', 'KRFT', 'KR', 'LB', 'LLL', 'LH', 'LRCX', 'LM', 'LEG', 'LEN', 'LVLT', 'LUK', 'LLY', 'LNC', 'LLTC', 'LMT', 'L', 'LOW', 'LYB', 'MTB', 'MAC', 'M', 'MNK', 'MRO', 'MPC', 'MAR', 'MMC', 'MLM', 'MAS', 'MA', 'MAT', 'MKC', 'MCD', 'MCK', 'MJN', 'MMV', 'MDT', 'MRK', 'MET', 'KORS', 'MCHP', 'MU', 'MSFT', 'MHK', 'TAP', 'MDLZ', 'MON', 'MNST', 'MCO', 'MS', 'MOS', 'MSI', 'MUR', 'MYL', 'NDAQ', 'NOV', 'NAVI', 'NTAP', 'NFLX', 'NWL', 'NFX', 'NEM', 'NWSA', 'NEE', 'NLSN', 'NKE', 'NI', 'NE', 'NBL', 'JWN', 'NSC', 'NTRS', 'NOC', 'NRG', 'NUE', 'NVDA', 'ORLY', 'OXY', 'OMC', 'OKE', 'ORCL', 'OI', 'PCAR', 'PLL', 'PH', 'PDCO', 'PAYX', 'PNR', 'PBCT', 'POM', 'PEP', 'PKI', 'PRGO', 'PFE', 'PCG', 'PM', 'PSX', 'PNW', 'PXD', 'PBI', 'PCL', 'PNC', 'RL', 'PPG', 'PPL', 'PX', 'PCP', 'PCLN', 'PFG', 'PG', 'PGR', 'PLD', 'PRU', 'PEG', 'PSA', 'PHM', 'PVH', 'QRVO', 'PWR', 'QCOM', 'DGX', 'RRC', 'RTN', 'O', 'RHT', 'REGN', 'RF', 'RSG', 'RAI', 'RHI', 'ROK', 'COL', 'ROP', 'ROST', 'RLD', 'R', 'CRM', 'SNDK', 'SCG', 'SLB', 'SNI', 'STX', 'SEE', 'SRE', 'SHW', 'SPG', 'SWKS', 'SLG', 'SJM', 'SNA', 'SO', 'LUV', 'SWN', 'SE', 'STJ', 'SWK', 'SPLS', 'SBUX', 'HOT', 'STT', 'SRCL', 'SYK', 'STI', 'SYMC', 'SYY', 'TROW', 'TGT', 'TEL', 'TE', 'TGNA', 'THC', 'TDC', 'TSO', 'TXN', 'TXT', 'HSY', 'TRV', 'TMO', 'TIF', 'TWX', 'TWC', 'TJX', 'TMK', 'TSS', 'TSCO', 'RIG', 'TRIP', 'FOXA', 'TSN', 'TYC', 'UA', 'UNP', 'UNH', 'UPS', 'URI', 'UTX', 'UHS', 'UNM', 'URBN', 'VFC', 'VLO', 'VAR', 'VTR', 'VRSN', 'VZ', 'VRTX', 'VIAB', 'V', 'VNO', 'VMC', 'WMT', 'WBA', 'DIS', 'WM', 'WAT', 'ANTM', 'WFC', 'WDC', 'WU', 'WY', 'WHR', 'WFM', 'WMB', 'WEC', 'WYN', 'WYNN', 'XEL', 'XRX', 'XLNX', 'XL', 'XYL', 'YHOO', 'YUM', 'ZBH', 'ZION', 'ZTS']

In [27]:
dfs = []
failed = []
for company in tqdm(sp500):
    time.sleep(np.random.uniform(0,2,1))
    try:
        comp = quandl_stocks(company)
        comp.columns = ['Open', 'High', 'Low', 'Close', 'Volume', 'Ex-Dividend', 'Split Ratio']
        comp['company_name'] = company
        dfs.append(comp)
    except:
        failed.append(company)
        print('Unable to scrape {}'.format(company))

 10%|█         | 51/496 [32:53<4:46:59, 38.70s/it]

Unable to scrape AVGO


 13%|█▎        | 63/496 [34:34<3:57:39, 32.93s/it]

Unable to scrape BRK.B


 15%|█▍        | 74/496 [35:52<3:24:34, 29.09s/it]

Unable to scrape BF.B


 30%|██▉       | 147/496 [51:42<2:02:45, 21.11s/it]

Unable to scrape DPS
Unable to scrape DTE
Unable to scrape DD
Unable to scrape DUK


 30%|███       | 151/496 [51:43<1:58:10, 20.55s/it]

Unable to scrape DNB


 31%|███       | 152/496 [51:44<1:57:06, 20.42s/it]

Unable to scrape ETFC


 31%|███       | 153/496 [51:45<1:56:02, 20.30s/it]

Unable to scrape EMN
Unable to scrape ETN


 31%|███▏      | 155/496 [51:46<1:53:54, 20.04s/it]

Unable to scrape EBAY
Unable to scrape ECL
Unable to scrape EIX


 32%|███▏      | 158/496 [51:47<1:50:47, 19.67s/it]

Unable to scrape EW
Unable to scrape EA


 32%|███▏      | 160/496 [51:48<1:48:48, 19.43s/it]

Unable to scrape EMC
Unable to scrape EMR


 33%|███▎      | 162/496 [51:49<1:46:51, 19.20s/it]

Unable to scrape ENDP


 33%|███▎      | 163/496 [51:50<1:45:54, 19.08s/it]

Unable to scrape ESV
Unable to scrape ETR
Unable to scrape EOG


 33%|███▎      | 166/496 [51:51<1:43:05, 18.74s/it]

Unable to scrape EQT


 44%|████▎     | 216/496 [1:23:18<1:47:59, 23.14s/it]

Unable to scrape HAR


 60%|█████▉    | 296/496 [1:34:42<1:03:59, 19.20s/it]

Unable to scrape MMV


 80%|████████  | 397/496 [2:08:59<32:10, 19.50s/it]  

Unable to scrape CRM


 94%|█████████▍| 468/496 [3:23:20<12:09, 26.07s/it]

Unable to scrape V


100%|██████████| 496/496 [3:29:34<00:00, 25.35s/it]


In [33]:
stocks = pd.concat(dfs)

In [36]:
stocks.to_csv('./stocks_cyc.csv')

# API failed to retrieve 27 of 500 stocks

In [38]:
print(failed)
print(len(failed))

['AVGO', 'BRK.B', 'BF.B', 'DPS', 'DTE', 'DD', 'DUK', 'DNB', 'ETFC', 'EMN', 'ETN', 'EBAY', 'ECL', 'EIX', 'EW', 'EA', 'EMC', 'EMR', 'ENDP', 'ESV', 'ETR', 'EOG', 'EQT', 'HAR', 'MMV', 'CRM', 'V']
27
