cleaning up week 37
This commit is contained in:
@@ -1485,7 +1485,7 @@ for polydegree in range(1, Maxpolydegree):
|
||||
trainingerror[polydegree] = 0.0
|
||||
for samples in range(trials):
|
||||
x_train, x_test, y_train, y_test = train_test_split(X, Energies, test_size=0.2)
|
||||
model = LinearRegression(fit_intercept=True).fit(x_train, y_train)
|
||||
model = LinearRegression(fit_intercept=False).fit(x_train, y_train)
|
||||
ypred = model.predict(x_train)
|
||||
ytilde = model.predict(x_test)
|
||||
testerror[polydegree] += mean_squared_error(y_test, ytilde)
|
||||
@@ -1506,10 +1506,12 @@ plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
Note that we kept the intercept column in the fitting here. This means that we need to set the _intercept_ in the call to the _Scikit-Learn_ function as _False_. Alternatively, we could have set up the design matrix $X$ without the first column of ones.
|
||||
|
||||
!split
|
||||
===== The same example but now with cross-validation =====
|
||||
|
||||
In this example we keep the intercept column again but add cross-validation in order to estimate the best possible value of the means squared error.
|
||||
!bc pycod
|
||||
# Common imports
|
||||
import os
|
||||
@@ -1567,7 +1569,7 @@ for polydegree in range(1, Maxpolydegree):
|
||||
polynomial[polydegree] = polydegree
|
||||
for degree in range(polydegree):
|
||||
X[:,degree] = Density**(degree/3.0)
|
||||
OLS = LinearRegression()
|
||||
OLS = LinearRegression(fit_intercept=False)
|
||||
# loop over trials in order to estimate the expectation value of the MSE
|
||||
estimated_mse_folds = cross_val_score(OLS, X, Energies, scoring='neg_mean_squared_error', cv=kfold)
|
||||
#[:, np.newaxis]
|
||||
@@ -1581,52 +1583,3 @@ plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== Cross-validation with Ridge =====
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user