cleaning up week 37
This commit is contained in:
@@ -1485,7 +1485,7 @@ for polydegree in range(1, Maxpolydegree):
|
||||
trainingerror[polydegree] = 0.0
|
||||
for samples in range(trials):
|
||||
x_train, x_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)
|
||||
model = LinearRegression(fit_intercept=True).fit(x_train, y_train)
|
||||
model = LinearRegression(fit_intercept=False).fit(x_train, y_train)
|
||||
ypred = model.predict(x_train)
|
||||
ytilde = model.predict(x_test)
|
||||
testerror[polydegree] += mean_squared_error(y_test, ytilde)
|
||||
@@ -1506,10 +1506,12 @@ plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
Note that we kept the intercept column in the fitting here. This means that we need to set the _intercept_ in the call to the _Scikit-Learn_ function as _False_. Alternatively, we could have set up the design matrix $X$ without the first column of ones.
|
||||
|
||||
!split
|
||||
===== The same example but now with cross-validation =====
|
||||
|
||||
In this example we keep the intercept column again but add cross-validation in order to estimate the best possible value of the means squared error.
|
||||
!bc pycod
|
||||
# Common imports
|
||||
import os
|
||||
@@ -1567,7 +1569,7 @@ for polydegree in range(1, Maxpolydegree):
|
||||
polynomial[polydegree] = polydegree
|
||||
for degree in range(polydegree):
|
||||
X[:,degree] = Density**(degree/3.0)
|
||||
OLS = LinearRegression()
|
||||
OLS = LinearRegression(fit_intercept=False)
|
||||
# loop over trials in order to estimate the expectation value of the MSE
|
||||
estimated_mse_folds = cross_val_score(OLS, X, Energies, scoring='neg_mean_squared_error', cv=kfold)
|
||||
#[:, np.newaxis]
|
||||
@@ -1581,52 +1583,3 @@ plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== Cross-validation with Ridge =====
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
import numpy as np
|
||||
from sklearn import datasets
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import GridSearchCV
|
||||
# load the diabetes datasets
|
||||
dataset = datasets.load_diabetes()
|
||||
# prepare a range of alpha values to test
|
||||
alphas = np.array([1,0.1,0.01,0.001,0.0001,0])
|
||||
# create and fit a ridge regression model, testing each alpha
|
||||
model = Ridge()
|
||||
grid = GridSearchCV(estimator=model, param_grid=dict(alpha=alphas))
|
||||
grid.fit(dataset.data, dataset.target)
|
||||
print(grid)
|
||||
# summarize the results of the grid search
|
||||
print(grid.best_score_)
|
||||
print(grid.best_estimator_.alpha)
|
||||
@@ -0,0 +1,17 @@
|
||||
import numpy as np
|
||||
from scipy.stats import uniform as sp_rand
|
||||
from sklearn import datasets
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import RandomizedSearchCV
|
||||
# load the diabetes datasets
|
||||
dataset = datasets.load_diabetes()
|
||||
# prepare a uniform distribution to sample for the alpha parameter
|
||||
param_grid = {'alpha': sp_rand()}
|
||||
# create and fit a ridge regression model, testing random alpha values
|
||||
model = Ridge()
|
||||
rsearch = RandomizedSearchCV(estimator=model, param_distributions=param_grid, n_iter=100)
|
||||
rsearch.fit(dataset.data, dataset.target)
|
||||
print(rsearch)
|
||||
# summarize the results of the random parameter search
|
||||
print(rsearch.best_score_)
|
||||
print(rsearch.best_estimator_.alpha)
|
||||
@@ -0,0 +1,30 @@
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
from sklearn.model_selection import GridSearchCV
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 10
|
||||
lambdas = np.logspace(-3, 3, nlambdas)
|
||||
|
||||
# create and fit a ridge regression model, testing each alpha
|
||||
model = Ridge()
|
||||
grid = GridSearchCV(estimator=model, param_grid=dict(alpha=lambdas))
|
||||
grid.fit(x, y)
|
||||
print(grid)
|
||||
# summarize the results of the grid search
|
||||
print(grid.best_score_)
|
||||
print(grid.best_estimator_.alpha)
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
#poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
@@ -0,0 +1,61 @@
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.model_selection import GridSearchCV
|
||||
|
||||
|
||||
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(315)
|
||||
|
||||
n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 5
|
||||
X = np.zeros((n,Maxpolydegree-1))
|
||||
|
||||
for degree in range(1,Maxpolydegree): #No intercept column
|
||||
X[:,degree-1] = x**(degree)
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
#Center by removing mean from each feature
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
|
||||
#Remove the intercept from the training data.
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
p = Maxpolydegree-1
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 10
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-4, 2, nlambdas)
|
||||
|
||||
# create and fit a ridge regression model, testing each alpha
|
||||
model = Ridge()
|
||||
grid = GridSearchCV(estimator=model, param_grid=dict(alpha=lambdas))
|
||||
grid.fit(X_train_scaled, y_train_scaled)
|
||||
print(grid)
|
||||
ypredictRidge = grid.predict(X_test_scaled)
|
||||
# summarize the results of the grid search
|
||||
print(grid.best_score_)
|
||||
print(grid.best_estimator_.alpha)
|
||||
print(MSE(y_test,ypredictRidge))
|
||||
@@ -0,0 +1,72 @@
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.model_selection import GridSearchCV
|
||||
from scipy.stats import uniform as sp_rand
|
||||
from sklearn.model_selection import RandomizedSearchCV
|
||||
|
||||
|
||||
def R2(y_data, y_model):
|
||||
return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
|
||||
|
||||
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(315)
|
||||
|
||||
n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 5
|
||||
X = np.zeros((n,Maxpolydegree-1))
|
||||
|
||||
for degree in range(1,Maxpolydegree): #No intercept column
|
||||
X[:,degree-1] = x**(degree)
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
#Center by removing mean from each feature
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
|
||||
#Remove the intercept from the training data.
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
y_test_scaled = y_test - y_scaler
|
||||
|
||||
p = Maxpolydegree-1
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 10
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
|
||||
param_grid = {'alpha': sp_rand()}
|
||||
# create and fit a ridge regression model, testing each alpha
|
||||
model = Ridge()
|
||||
grid = RandomizedSearchCV(estimator=model, param_distributions=param_grid, n_iter=100)
|
||||
#GridSearchCV(estimator=model, param_grid=dict(alpha=lambdas))
|
||||
grid.fit(X_train_scaled, y_train_scaled)
|
||||
print(grid)
|
||||
ypredictRidge = grid.predict(X_test_scaled)
|
||||
# summarize the results of the grid search
|
||||
print(grid.best_score_)
|
||||
print(grid.best_estimator_.alpha)
|
||||
print(MSE(y_test_scaled,ypredictRidge))
|
||||
print(R2(y_test_scaled,ypredictRidge))
|
||||
|
||||
|
||||
|
||||
@@ -31,41 +31,6 @@ in evaluating the MSE as function of different $\lambda$ values.
|
||||
Based on these calculations, one tries then to determine the value of the hyperparameter $\lambda$
|
||||
which results in optimal scores (for example the smallest MSE or an $R2=1$).
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
We see from this plot that the optimal MSE occurs for a value of $\lambda\in [10,100]$.
|
||||
@@ -80,41 +45,7 @@ included with the library _Scikit-Learn_, as demonstrated for the same
|
||||
example here.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user