# Chapter 15: Multiple Regression

In [14]:
from __future__ import division
import numpy as np
import random

In [15]:
%%capture
%run -i 'gradient_descent.ipynb' #minimize_stochastic

## The model

$$y_i = \alpha + \beta_1x_{i1} + \ldots + \beta_kx_{ik}$$ 

In [3]:
def predict(x_i, beta):
    """assumes the first element of each x_i is 1"""
    return np.dot(x_i, beta)

### Data

$x_i$ is of the form <code>[constant term, number of friends, work hours per day, 1 if has phd otherwise 0]</code>

In [4]:
x = [[1,49,4,0],[1,41,9,0],[1,40,8,0],[1,25,6,0],[1,21,1,0],[1,21,0,0],[1,19,3,0],[1,19,0,0],[1,18,9,0],[1,18,8,0],[1,16,4,0],[1,15,3,0],[1,15,0,0],[1,15,2,0],[1,15,7,0],[1,14,0,0],[1,14,1,0],[1,13,1,0],[1,13,7,0],[1,13,4,0],[1,13,2,0],[1,12,5,0],[1,12,0,0],[1,11,9,0],[1,10,9,0],[1,10,1,0],[1,10,1,0],[1,10,7,0],[1,10,9,0],[1,10,1,0],[1,10,6,0],[1,10,6,0],[1,10,8,0],[1,10,10,0],[1,10,6,0],[1,10,0,0],[1,10,5,0],[1,10,3,0],[1,10,4,0],[1,9,9,0],[1,9,9,0],[1,9,0,0],[1,9,0,0],[1,9,6,0],[1,9,10,0],[1,9,8,0],[1,9,5,0],[1,9,2,0],[1,9,9,0],[1,9,10,0],[1,9,7,0],[1,9,2,0],[1,9,0,0],[1,9,4,0],[1,9,6,0],[1,9,4,0],[1,9,7,0],[1,8,3,0],[1,8,2,0],[1,8,4,0],[1,8,9,0],[1,8,2,0],[1,8,3,0],[1,8,5,0],[1,8,8,0],[1,8,0,0],[1,8,9,0],[1,8,10,0],[1,8,5,0],[1,8,5,0],[1,7,5,0],[1,7,5,0],[1,7,0,0],[1,7,2,0],[1,7,8,0],[1,7,10,0],[1,7,5,0],[1,7,3,0],[1,7,3,0],[1,7,6,0],[1,7,7,0],[1,7,7,0],[1,7,9,0],[1,7,3,0],[1,7,8,0],[1,6,4,0],[1,6,6,0],[1,6,4,0],[1,6,9,0],[1,6,0,0],[1,6,1,0],[1,6,4,0],[1,6,1,0],[1,6,0,0],[1,6,7,0],[1,6,0,0],[1,6,8,0],[1,6,4,0],[1,6,2,1],[1,6,1,1],[1,6,3,1],[1,6,6,1],[1,6,4,1],[1,6,4,1],[1,6,1,1],[1,6,3,1],[1,6,4,1],[1,5,1,1],[1,5,9,1],[1,5,4,1],[1,5,6,1],[1,5,4,1],[1,5,4,1],[1,5,10,1],[1,5,5,1],[1,5,2,1],[1,5,4,1],[1,5,4,1],[1,5,9,1],[1,5,3,1],[1,5,10,1],[1,5,2,1],[1,5,2,1],[1,5,9,1],[1,4,8,1],[1,4,6,1],[1,4,0,1],[1,4,10,1],[1,4,5,1],[1,4,10,1],[1,4,9,1],[1,4,1,1],[1,4,4,1],[1,4,4,1],[1,4,0,1],[1,4,3,1],[1,4,1,1],[1,4,3,1],[1,4,2,1],[1,4,4,1],[1,4,4,1],[1,4,8,1],[1,4,2,1],[1,4,4,1],[1,3,2,1],[1,3,6,1],[1,3,4,1],[1,3,7,1],[1,3,4,1],[1,3,1,1],[1,3,10,1],[1,3,3,1],[1,3,4,1],[1,3,7,1],[1,3,5,1],[1,3,6,1],[1,3,1,1],[1,3,6,1],[1,3,10,1],[1,3,2,1],[1,3,4,1],[1,3,2,1],[1,3,1,1],[1,3,5,1],[1,2,4,1],[1,2,2,1],[1,2,8,1],[1,2,3,1],[1,2,1,1],[1,2,9,1],[1,2,10,1],[1,2,9,1],[1,2,4,1],[1,2,5,1],[1,2,0,1],[1,2,9,1],[1,2,9,1],[1,2,0,1],[1,2,1,1],[1,2,1,1],[1,2,4,1],[1,1,0,1],[1,1,2,1],[1,1,2,1],[1,1,5,1],[1,1,3,1],[1,1,10,1],[1,1,6,1],[1,1,0,1],[1,1,8,1],[1,1,6,1],[1,1,4,1],[1,1,9,1],[1,1,9,1],[1,1,4,1],[1,1,2,1],[1,1,9,1],[1,1,0,1],[1,1,8,1],[1,1,6,1],[1,1,1,1],[1,1,1,1],[1,1,5,1]]
daily_minutes_good = [68.77,51.25,52.08,38.36,44.54,57.13,51.4,41.42,31.22,34.76,54.01,38.79,47.59,49.1,27.66,41.03,36.73,48.65,28.12,46.62,35.57,32.98,35,26.07,23.77,39.73,40.57,31.65,31.21,36.32,20.45,21.93,26.02,27.34,23.49,46.94,30.5,33.8,24.23,21.4,27.94,32.24,40.57,25.07,19.42,22.39,18.42,46.96,23.72,26.41,26.97,36.76,40.32,35.02,29.47,30.2,31,38.11,38.18,36.31,21.03,30.86,36.07,28.66,29.08,37.28,15.28,24.17,22.31,30.17,25.53,19.85,35.37,44.6,17.23,13.47,26.33,35.02,32.09,24.81,19.33,28.77,24.26,31.98,25.73,24.86,16.28,34.51,15.23,39.72,40.8,26.06,35.76,34.76,16.13,44.04,18.03,19.65,32.62,35.59,39.43,14.18,35.24,40.13,41.82,35.45,36.07,43.67,24.61,20.9,21.9,18.79,27.61,27.21,26.61,29.77,20.59,27.53,13.82,33.2,25,33.1,36.65,18.63,14.87,22.2,36.81,25.53,24.62,26.25,18.21,28.08,19.42,29.79,32.8,35.99,28.32,27.79,35.88,29.06,36.28,14.1,36.63,37.49,26.9,18.58,38.48,24.48,18.95,33.55,14.24,29.04,32.51,25.63,22.22,19,32.73,15.16,13.9,27.2,32.01,29.27,33,13.74,20.42,27.32,18.23,35.35,28.48,9.08,24.62,20.12,35.26,19.92,31.02,16.49,12.16,30.7,31.22,34.65,13.13,27.51,33.2,31.57,14.1,33.42,17.44,10.12,24.42,9.82,23.39,30.93,15.03,21.67,31.09,33.29,22.61,26.89,23.48,8.38,27.81,32.35,23.84]

## Fitting the Model

In [5]:
def error(x_i, y_i, beta):
    return y_i - predict(x_i, beta)

In [6]:
def squared_error(x_i, y_i, beta):
    return error(x_i, y_i, beta) ** 2

In [7]:
def squared_error_gradient(x_i, y_i, beta):
    """the gradient corresponding to the ith squared error term"""
    return [-2 * x_ij * error(x_i, y_i, beta)
            for x_ij in x_i]

In [8]:
def estimate_beta(x, y):
    beta_initial = [random.random() for x_i in x[0]]
    return minimize_stochastic(squared_error, squared_error_gradient, x, y, beta_initial, 0.001)

In [9]:
random.seed(0)
estimate_beta(x, daily_minutes_good)

[30.625234786488353,
 0.97154482886965354,
 -1.8679272872032218,
 0.91145694992144499]

## Goodness of Fit

In [12]:
def total_sum_of_squares(y):
    """the total squared variation for y_i's from their mean"""
    mean = np.mean(y)
    de_meaned = [y_i - mean for y_i in y]
    return sum(v ** 2 for v in de_meaned)

In [10]:
def multiple_r_squared(x, y, beta):
    sum_of_squared_errors = sum(error(x_i, y_i, beta) ** 2 for x_i, y_i in zip(x, y))
    return 1.0 - sum_of_squared_errors / total_sum_of_squares(y)

In [13]:
beta = estimate_beta(x, daily_minutes_good)
multiple_r_squared(x, daily_minutes_good, beta)

0.68000494205037876

## Digression: The Bootstrap

In [17]:
def bootstrap_sample(data):
    """randomly samples len(data) elements with replacement"""
    return [random.choice(data) for _ in data]

In [19]:
def bootstrap_statistic(data, stats_fn, num_samples):
    """evaluates stats_fn on num_samples bootstrap samples from data"""
    return [stats_fn(bootstrap_sample(data)) for _ in range(num_samples)]

In [20]:
# 101 points all very close to 100
close_to_100 = [99.5 + random.random() for _ in range(101)]
# 101 points, 50 of them near 0, 50 of them near 200
far_from_100 = ([99.5 + random.random()] + [random.random() for _ in range(50)] + [200 + random.random() for _ in range(50)])

In [21]:
np.median(close_to_100)

100.01503949815732

In [22]:
np.median(far_from_100)

100.23314450795888

In [23]:
bootstrap_statistic(close_to_100, np.median, 100)

[100.10659139884376,
 100.00861843623038,
 99.930512889395629,
 100.05678451489779,
 99.930512889395629,
 100.06692142722724,
 100.05678451489779,
 100.0798666333307,
 100.04842178962554,
 100.00604808956813,
 100.04842178962554,
 100.01173608665734,
 100.00909588188595,
 100.01173608665734,
 100.01503949815732,
 99.923050180023694,
 100.01173608665734,
 100.02659145198471,
 100.00861843623038,
 100.01173608665734,
 100.05919390133619,
 100.06692142722724,
 100.04842178962554,
 100.02659145198471,
 100.01290982599281,
 100.00604808956813,
 100.04842178962554,
 99.936327437298146,
 100.01290982599281,
 100.02659145198471,
 100.02659145198471,
 100.01290982599281,
 100.00909588188595,
 100.04842178962554,
 100.01173608665734,
 100.02659145198471,
 99.927718845246019,
 100.02479234638265,
 100.01503949815732,
 100.02659145198471,
 100.02659145198471,
 100.01173608665734,
 100.01290982599281,
 100.05719883816568,
 100.02659145198471,
 100.06692142722724,
 100.02479234638265,
 100.098174293

In [25]:
bootstrap_statistic(far_from_100, np.median, 100)

[0.8144765899513936,
 0.80670323630407359,
 0.80670323630407359,
 200.17840484893861,
 200.02326439867943,
 0.93612418085481397,
 200.02435474865004,
 200.11109237050712,
 100.23314450795888,
 0.93612418085481397,
 200.08881054316973,
 0.81210435036957829,
 200.124738575685,
 200.17840484893861,
 200.06881195293602,
 200.02326439867943,
 0.80670323630407359,
 200.13119908834028,
 0.93612418085481397,
 0.90668644017511935,
 200.06881195293602,
 200.124738575685,
 200.124738575685,
 0.90668644017511935,
 200.06881195293602,
 200.124738575685,
 0.94791797327840432,
 0.93612418085481397,
 0.83522619971043122,
 0.95009424239637119,
 200.13119908834028,
 0.95009424239637119,
 200.08881054316973,
 200.02326439867943,
 200.11109237050712,
 200.124738575685,
 100.23314450795888,
 200.19672584917518,
 200.17840484893861,
 0.99128726827273905,
 0.93612418085481397,
 100.23314450795888,
 0.99128726827273905,
 0.81210435036957829,
 0.94791797327840432,
 0.83522619971043122,
 0.99128726827273905,
 2