# **Linear Regression**
---
 
- Copyright (c) Lukas Gonon, 2024. All rights reserved

- Author: Lukas Gonon <l.gonon@imperial.ac.uk>

- Platform: Tested on Windows 10 with Python 3.9

In [None]:
import numpy as np
import pandas as pd
pd.core.common.is_list_like = pd.api.types.is_list_like
from datetime import datetime
%matplotlib inline
import matplotlib.pyplot as plt
import random
import yahoo_fin.stock_info as yf
import statsmodels.api as sm

#### List of tickers

In [None]:
tickers = ['MMM', 'ABT', 'ABBV', 'ACN', 'ATVI', 'AYI', 'ADBE', 'AMD', 'AAP', 'AES', 'AET', 'AMG', 'AFL', 'A', 'APD', 'AKAM', 'ALK', 'ALB', 'ARE', 'ALXN', 'ALGN', 'ALLE', 'AGN', 'ADS', 'LNT', 'ALL', 'GOOGL', 'GOOG', 'MO', 'AMZN', 'AEE', 'AAL', 'AEP', 'AXP', 'AIG', 'AMT', 'AWK', 'AMP', 'ABC', 'AME', 'AMGN', 'APH', 'APC', 'ADI', 'ANDV', 'ANSS', 'ANTM', 'AON', 'AOS', 'APA', 'AIV', 'AAPL', 'AMAT', 'APTV', 'ADM', 'ARNC', 'AJG', 'AIZ', 'T', 'ADSK', 'ADP', 'AZO', 'AVB', 'AVY', 'BHGE', 'BLL', 'BAC', 'BK', 'BAX', 'BBT', 'BDX', 'BRK.B', 'BBY', 'BIIB', 'BLK', 'HRB', 'BA', 'BWA', 'BXP', 'BSX', 'BHF', 'BMY', 'AVGO', 'BF.B', 'CHRW', 'CA', 'COG', 'CDNS', 'CPB', 'COF', 'CAH', 'CBOE', 'KMX', 'CCL', 'CAT', 'CBG', 'CBS', 'CELG', 'CNC', 'CNP', 'CTL', 'CERN', 'CF', 'SCHW', 'CHTR', 'CHK', 'CVX', 'CMG', 'CB', 'CHD', 'CI', 'XEC', 'CINF', 'CTAS', 'CSCO', 'C', 'CFG', 'CTXS', 'CLX', 'CME', 'CMS', 'KO', 'CTSH', 'CL', 'CMCSA', 'CMA', 'CAG', 'CXO', 'COP', 'ED', 'STZ', 'COO', 'GLW', 'COST', 'COTY', 'CCI', 'CSRA', 'CSX', 'CMI', 'CVS', 'DHI', 'DHR', 'DRI', 'DVA', 'DE', 'DAL', 'XRAY', 'DVN', 'DLR', 'DFS', 'DISCA', 'DISCK', 'DISH', 'DG', 'DLTR', 'D', 'DOV', 'DWDP', 'DPS', 'DTE', 'DRE', 'DUK', 'DXC', 'ETFC', 'EMN', 'ETN', 'EBAY', 'ECL', 'EIX', 'EW', 'EA', 'EMR', 'ETR', 'EVHC', 'EOG', 'EQT', 'EFX', 'EQIX', 'EQR', 'ESS', 'EL', 'ES', 'RE', 'EXC', 'EXPE', 'EXPD', 'ESRX', 'EXR', 'XOM', 'FFIV', 'FB', 'FAST', 'FRT', 'FDX', 'FIS', 'FITB', 'FE', 'FISV', 'FLIR', 'FLS', 'FLR', 'FMC', 'FL', 'F', 'FTV', 'FBHS', 'BEN', 'FCX', 'GPS', 'GRMN', 'IT', 'GD', 'GE', 'GGP', 'GIS', 'GM', 'GPC', 'GILD', 'GPN', 'GS', 'GT', 'GWW', 'HAL', 'HBI', 'HOG', 'HRS', 'HIG', 'HAS', 'HCA', 'HCP', 'HP', 'HSIC', 'HSY', 'HES', 'HPE', 'HLT', 'HOLX', 'HD', 'HON', 'HRL', 'HST', 'HPQ', 'HUM', 'HBAN', 'HII', 'IDXX', 'INFO', 'ITW', 'ILMN', 'IR', 'INTC', 'ICE', 'IBM', 'INCY', 'IP', 'IPG', 'IFF', 'INTU', 'ISRG', 'IVZ', 'IQV', 'IRM', 'JEC', 'JBHT', 'SJM', 'JNJ', 'JCI', 'JPM', 'JNPR', 'KSU', 'K', 'KEY', 'KMB', 'KIM', 'KMI', 'KLAC', 'KSS', 'KHC', 'KR', 'LB', 'LLL', 'LH', 'LRCX', 'LEG', 'LEN', 'LUK', 'LLY', 'LNC', 'LKQ', 'LMT', 'L', 'LOW', 'LYB', 'MTB', 'MAC', 'M', 'MRO', 'MPC', 'MAR', 'MMC', 'MLM', 'MAS', 'MA', 'MAT', 'MKC', 'MCD', 'MCK', 'MDT', 'MRK', 'MET', 'MTD', 'MGM', 'KORS', 'MCHP', 'MU', 'MSFT', 'MAA', 'MHK', 'TAP', 'MDLZ', 'MON', 'MNST', 'MCO', 'MS', 'MOS', 'MSI', 'MYL', 'NDAQ', 'NOV', 'NAVI', 'NTAP', 'NFLX', 'NWL', 'NFX', 'NEM', 'NWSA', 'NWS', 'NEE', 'NLSN', 'NKE', 'NI', 'NBL', 'JWN', 'NSC', 'NTRS', 'NOC', 'NCLH', 'NRG', 'NUE', 'NVDA', 'ORLY', 'OXY', 'OMC', 'OKE', 'ORCL', 'PCAR', 'PKG', 'PH', 'PDCO', 'PAYX', 'PYPL', 'PNR', 'PBCT', 'PEP', 'PKI', 'PRGO', 'PFE', 'PCG', 'PM', 'PSX', 'PNW', 'PXD', 'PNC', 'RL', 'PPG', 'PPL', 'PX', 'PCLN', 'PFG', 'PG', 'PGR', 'PLD', 'PRU', 'PEG', 'PSA', 'PHM', 'PVH', 'QRVO', 'PWR', 'QCOM', 'DGX', 'RRC', 'RJF', 'RTN', 'O', 'RHT', 'REG', 'REGN', 'RF', 'RSG', 'RMD', 'RHI', 'ROK', 'COL', 'ROP', 'ROST', 'RCL', 'CRM', 'SBAC', 'SCG', 'SLB', 'SNI', 'STX', 'SEE', 'SRE', 'SHW', 'SIG', 'SPG', 'SWKS', 'SLG', 'SNA', 'SO', 'LUV', 'SPGI', 'SWK', 'SBUX', 'STT', 'SRCL', 'SYK', 'STI', 'SYMC', 'SYF', 'SNPS', 'SYY', 'TROW', 'TPR', 'TGT', 'TEL', 'FTI', 'TXN', 'TXT', 'TMO', 'TIF', 'TWX', 'TJX', 'TMK', 'TSS', 'TSCO', 'TDG', 'TRV', 'TRIP', 'FOXA', 'FOX', 'TSN', 'UDR', 'ULTA', 'USB', 'UAA', 'UA', 'UNP', 'UAL', 'UNH', 'UPS', 'URI', 'UTX', 'UHS', 'UNM', 'VFC', 'VLO', 'VAR', 'VTR', 'VRSN', 'VRSK', 'VZ', 'VRTX', 'VIAB', 'V', 'VNO', 'VMC', 'WMT', 'WBA', 'DIS', 'WM', 'WAT', 'WEC', 'WFC', 'HCN', 'WDC', 'WU', 'WRK', 'WY', 'WHR', 'WMB', 'WLTW', 'WYN', 'WYNN', 'XEL', 'XRX', 'XLNX', 'XL', 'XYL', 'YUM', 'ZBH', 'ZION', 'ZTS']

In [None]:
data_source = 'yahoo'
start, end = '2016-01-01', '2018-01-01'

ticker = 'SPY'
df = yf.get_data(ticker,start, end)
df.head()

df = df.drop(columns=['open', 'high', 'low', 'adjclose', 'volume','ticker'])
df.columns=[ticker]


In [None]:
nbTickers = len(tickers)
nbExtract = 7 ## Number of tickers to consider
listExtractTickers = [ticker]
listTemp = random.sample(tickers, nbExtract)
for i in range(nbExtract):
    listExtractTickers.append(listTemp[i]) ## select nbExtract out of the whole list
print("List of extracted tickers: ", listExtractTickers)

In [None]:
## We construct a DataFrame with the first 5 tickers only
for ticker in listExtractTickers[1:]:
    df0 = yf.get_data(ticker, start, end)
    df0 = df0.drop(columns=['open', 'high', 'low', 'adjclose', 'volume','ticker'])
    df0.columns=[ticker]
    df = pd.concat([df, df0], axis=1)

In [None]:
print("Total number of tickers:", nbTickers)
print("Numer of observations:", len(df))
print("Number of tickers selected:", nbExtract)
df.head()

Compute the returns

In [None]:
returns = df.pct_change(1)
returns= returns.dropna()
returns.head()

### Scatter plot

In [None]:
nbCols = 3
nbRows = 1+nbExtract//nbCols
rem = nbExtract%nbCols
fig, axs = plt.subplots(nrows=nbRows, ncols=nbCols, figsize=(20, 4*nbRows))
for i in range(nbRows):
    for j in range(0,nbCols):
        if (nbCols*i+j)<nbExtract:
            plt.subplot(nbRows, nbCols, nbCols*i+j+1)
            idCol = returns.columns[3*i+1]
            plt.scatter(returns[idCol], returns["SPY"])
            plt.ylabel("SPY", size=15)
            plt.xlabel("%s (ID:%i)" %(idCol, nbCols*i+j), size=15)
plt.subplots_adjust(hspace=0.3, wspace=1.)
plt.show()

### Linear regression of SPY against one of the components

#### Choice of the component

In [None]:
choiceIndex = 5
stockLabel = returns.columns[choiceIndex]
plt.figure(figsize = (6,4))
idCol = returns.columns[choiceIndex]
plt.scatter(returns[stockLabel], returns["SPY"])
plt.ylabel("SPY", size=15)
plt.xlabel("%s (ID:%i)" %(idCol, choiceIndex), size=15)
plt.show()

#### Run a linear regression

In [None]:
X = returns[stockLabel].values
Y = returns["SPY"].values
modle = sm.add_constant(X)
model = sm.OLS(Y, X).fit()
print(model.summary())

In [None]:
print('parameters: ', model.params)

In [None]:
plt.figure(figsize = (6, 4))
plt.scatter(returns[stockLabel], returns["SPY"])
plt.ylabel('SPY', size=15)
plt.xlabel(stockLabel, size=15)
plt.title("Linear regression results", size=15)
plt.plot(returns[stockLabel], model.predict(),'r-')
plt.show()

### Does adding variables increase the $R^2$?

In [None]:
rSquared = []
N = 1
regressionTickers = returns.columns[1]
X = returns[regressionTickers].values
Y = returns["SPY"].values
X = sm.add_constant(X)
model = sm.OLS(Y, X).fit()
rSquared.append(model.rsquared)

for N in range(2, nbExtract):
    regressionTickers = [returns.columns[i] for i in range(1, N)]
    X = np.column_stack((returns[rt].values for rt in regressionTickers))
    Y = returns["SPY"].values
    X = sm.add_constant(X)
    model = sm.OLS(Y, X).fit()
    rSquared.append(model.rsquared)

In [None]:
df = pd.DataFrame({ 'Nb Regressors' : range(1,nbExtract),'RSquared' : rSquared})
df = df.set_index('Nb Regressors')
df['RSquared'].plot();

#### Linear Regression with all the factors

In [None]:
N = nbExtract
regressionTickers = [returns.columns[i] for i in range(1, N)]
X = np.column_stack((returns[rt].values for rt in regressionTickers))
Y = returns["SPY"].values
X = sm.add_constant(X)
model = sm.OLS(Y, X).fit()
print(model.summary())

### Linear Regression using Scikit Learn

In [None]:
from sklearn.linear_model import LinearRegression
x, y = returns[stockLabel].values, returns["SPY"].values
length = len(returns[stockLabel].values)
x = x.reshape(length, 1)
y = y.reshape(length, 1)
regr = LinearRegression()
regr.fit(x, y)

In [None]:
plt.scatter(y, x,  color='blue')
plt.plot(regr.predict(x), x, color='red', linewidth=2)
plt.xlabel(stockLabel, size=15)
plt.ylabel("SPY", size=15)
plt.show()

In [None]:
print("Regression coefficient: ", regr.coef_)
print("Regression intercept: ", regr.intercept_)