update week39
This commit is contained in:
@@ -23,13 +23,99 @@ For a good discussion on gradient methods, see Goodfellow et al section 4.3-4.5
|
||||
!split
|
||||
===== Searching for Optimal Regularization Parameters $\lambda$ =====
|
||||
|
||||
In project 1 when using Ridge and Lasso regression, we end up
|
||||
In project 1, when using Ridge and Lasso regression, we end up
|
||||
searching for the optimal parameter $\lambda$ which minimizes our
|
||||
selected scores (MSE or $R2$ values for example). The brute force
|
||||
approach, as discussed in the code here for Ridge regression consists
|
||||
approach, as discussed in the code here for Ridge regression, consists
|
||||
in evaluating the MSE as function of different $\lambda$ values.
|
||||
Based on these calculations, one tries then to determine the value of the hyperparameter $\lambda$
|
||||
which results in optimal scores (for example the smallest MSE or an $R2=1$).
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
An alternative is to use the so-called grid search functionality included with the library _Scikit-Learn_, as demonstrated for the same example here.
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
We see from this plot that the optimal MSE occurs for a value of $\lambda\in [10,100]$.
|
||||
In order to nail down the best value of $\lambda$, we could in turn narrow down the search area.
|
||||
|
||||
!split
|
||||
===== Grid Search =====
|
||||
|
||||
|
||||
An alternative is to use the so-called grid search functionality
|
||||
included with the library _Scikit-Learn_, as demonstrated for the same
|
||||
example here.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
@@ -1290,3 +1376,5 @@ plt.show()
|
||||
_Challenge_: try to write a similar code for a Logistic Regression case.
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user