update week39
This commit is contained in:
@@ -47,6 +47,7 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'searching-for-optimal-regularization-parameters-lambda'),
|
||||
('Grid Search', 2, None, 'grid-search'),
|
||||
('Optimization, the central part of any Machine Learning '
|
||||
'algortithm',
|
||||
2,
|
||||
@@ -216,56 +217,57 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs001.html#plan-for-week-39" style="font-size: 80%;">Plan for week 39</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs002.html#thursday-september-30" style="font-size: 80%;">Thursday September 30</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs003.html#searching-for-optimal-regularization-parameters-lambda" style="font-size: 80%;">Searching for Optimal Regularization Parameters \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs004.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs005.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs006.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs007.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs008.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs009.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs010.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs011.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs012.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs013.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs014.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs015.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs016.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs017.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs018.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs019.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs020.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs021.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs022.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs024.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs024.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs025.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs026.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs031.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs034.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#revisiting-our-first-homework" style="font-size: 80%;">Revisiting our first homework</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs040.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs037.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs038.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs039.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs040.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs041.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs042.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs043.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs044.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs045.html#friday-october-1" style="font-size: 80%;">Friday October 1</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs046.html#stochastic-gradient-descent" style="font-size: 80%;">Stochastic Gradient Descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs047.html#computation-of-gradients" style="font-size: 80%;">Computation of gradients</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs048.html#sgd-example" style="font-size: 80%;">SGD example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs049.html#the-gradient-step" style="font-size: 80%;">The gradient step</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs050.html#simple-example-code" style="font-size: 80%;">Simple example code</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs051.html#when-do-we-stop" style="font-size: 80%;">When do we stop?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs052.html#slightly-different-approach" style="font-size: 80%;">Slightly different approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs053.html#program-for-stochastic-gradient" style="font-size: 80%;">Program for stochastic gradient</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs004.html#grid-search" style="font-size: 80%;">Grid Search</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs005.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs006.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs007.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs008.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs009.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs010.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs011.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs012.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs013.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs014.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs015.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs016.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs017.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs018.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs019.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs020.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs021.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs022.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs023.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs025.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs025.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs026.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs027.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs032.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs035.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs036.html#revisiting-our-first-homework" style="font-size: 80%;">Revisiting our first homework</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs041.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs038.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs039.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs040.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs041.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs042.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs043.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs044.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs045.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs046.html#friday-october-1" style="font-size: 80%;">Friday October 1</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs047.html#stochastic-gradient-descent" style="font-size: 80%;">Stochastic Gradient Descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs048.html#computation-of-gradients" style="font-size: 80%;">Computation of gradients</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs049.html#sgd-example" style="font-size: 80%;">SGD example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs050.html#the-gradient-step" style="font-size: 80%;">The gradient step</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs051.html#simple-example-code" style="font-size: 80%;">Simple example code</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs052.html#when-do-we-stop" style="font-size: 80%;">When do we stop?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs053.html#slightly-different-approach" style="font-size: 80%;">Slightly different approach</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week39-bs054.html#program-for-stochastic-gradient" style="font-size: 80%;">Program for stochastic gradient</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -324,7 +326,7 @@ MathJax.Hub.Config({
|
||||
<li><a href="._week39-bs008.html">9</a></li>
|
||||
<li><a href="._week39-bs009.html">10</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week39-bs053.html">54</a></li>
|
||||
<li><a href="._week39-bs054.html">55</a></li>
|
||||
<li><a href="._week39-bs001.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -185,14 +185,105 @@ For a good discussion on gradient methods, see Goodfellow et al section 4.3-4.5
|
||||
<h2 id="searching-for-optimal-regularization-parameters-lambda">Searching for Optimal Regularization Parameters \( \lambda \) </h2>
|
||||
|
||||
<p>
|
||||
In project 1 when using Ridge and Lasso regression, we end up
|
||||
In project 1, when using Ridge and Lasso regression, we end up
|
||||
searching for the optimal parameter \( \lambda \) which minimizes our
|
||||
selected scores (MSE or \( R2 \) values for example). The brute force
|
||||
approach, as discussed in the code here for Ridge regression consists
|
||||
approach, as discussed in the code here for Ridge regression, consists
|
||||
in evaluating the MSE as function of different \( \lambda \) values.
|
||||
Based on these calculations, one tries then to determine the value of the hyperparameter \( \lambda \)
|
||||
which results in optimal scores (for example the smallest MSE or an \( R2=1 \)).
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> KFold
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> Ridge
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> cross_val_score
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> PolynomialFeatures
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
np.random.seed(<span style="color: #B452CD">3155</span>)
|
||||
<span style="color: #228B22"># Generate the data.</span>
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.linspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">3</span>, n).reshape(-<span style="color: #B452CD">1</span>, <span style="color: #B452CD">1</span>)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)+ np.random.normal(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">0.1</span>, x.shape)
|
||||
<span style="color: #228B22"># Decide degree on polynomial to fit</span>
|
||||
poly = PolynomialFeatures(degree = <span style="color: #B452CD">10</span>)
|
||||
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">500</span>
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">5</span>, nlambdas)
|
||||
<span style="color: #228B22"># Initialize a KFold instance</span>
|
||||
k = <span style="color: #B452CD">5</span>
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = <span style="color: #B452CD">0</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> lmb <span style="color: #8B008B">in</span> lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring=<span style="color: #CD5555">'neg_mean_squared_error'</span>, cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += <span style="color: #B452CD">1</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = <span style="color: #CD5555">'cross_val_score'</span>)
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We see from this plot that the optimal MSE occurs for a value of \( \lambda\in [10,100] \).
|
||||
In order to nail down the best value of \( \lambda \), we could in turn narrow down the search area.
|
||||
</section>
|
||||
|
||||
|
||||
<section>
|
||||
<h2 id="grid-search">Grid Search </h2>
|
||||
|
||||
<p>
|
||||
An alternative is to use the so-called grid search functionality included with the library <b>Scikit-Learn</b>, as demonstrated for the same example here.
|
||||
An alternative is to use the so-called grid search functionality
|
||||
included with the library <b>Scikit-Learn</b>, as demonstrated for the same
|
||||
example here.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> KFold
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> Ridge
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> cross_val_score
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> PolynomialFeatures
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
np.random.seed(<span style="color: #B452CD">3155</span>)
|
||||
<span style="color: #228B22"># Generate the data.</span>
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.linspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">3</span>, n).reshape(-<span style="color: #B452CD">1</span>, <span style="color: #B452CD">1</span>)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)+ np.random.normal(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">0.1</span>, x.shape)
|
||||
<span style="color: #228B22"># Decide degree on polynomial to fit</span>
|
||||
poly = PolynomialFeatures(degree = <span style="color: #B452CD">10</span>)
|
||||
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">500</span>
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">5</span>, nlambdas)
|
||||
<span style="color: #228B22"># Initialize a KFold instance</span>
|
||||
k = <span style="color: #B452CD">5</span>
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = <span style="color: #B452CD">0</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> lmb <span style="color: #8B008B">in</span> lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring=<span style="color: #CD5555">'neg_mean_squared_error'</span>, cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += <span style="color: #B452CD">1</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = <span style="color: #CD5555">'cross_val_score'</span>)
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
</section>
|
||||
|
||||
|
||||
|
||||
@@ -67,6 +67,7 @@ div { text-align: justify; text-justify: inter-word; }
|
||||
2,
|
||||
None,
|
||||
'searching-for-optimal-regularization-parameters-lambda'),
|
||||
('Grid Search', 2, None, 'grid-search'),
|
||||
('Optimization, the central part of any Machine Learning '
|
||||
'algortithm',
|
||||
2,
|
||||
@@ -267,15 +268,105 @@ For a good discussion on gradient methods, see Goodfellow et al section 4.3-4.5
|
||||
<h2 id="searching-for-optimal-regularization-parameters-lambda">Searching for Optimal Regularization Parameters \( \lambda \) </h2>
|
||||
|
||||
<p>
|
||||
In project 1 when using Ridge and Lasso regression, we end up
|
||||
In project 1, when using Ridge and Lasso regression, we end up
|
||||
searching for the optimal parameter \( \lambda \) which minimizes our
|
||||
selected scores (MSE or \( R2 \) values for example). The brute force
|
||||
approach, as discussed in the code here for Ridge regression consists
|
||||
approach, as discussed in the code here for Ridge regression, consists
|
||||
in evaluating the MSE as function of different \( \lambda \) values.
|
||||
Based on these calculations, one tries then to determine the value of the hyperparameter \( \lambda \)
|
||||
which results in optimal scores (for example the smallest MSE or an \( R2=1 \)).
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> KFold
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> Ridge
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> cross_val_score
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> PolynomialFeatures
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
np.random.seed(<span style="color: #B452CD">3155</span>)
|
||||
<span style="color: #228B22"># Generate the data.</span>
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.linspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">3</span>, n).reshape(-<span style="color: #B452CD">1</span>, <span style="color: #B452CD">1</span>)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)+ np.random.normal(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">0.1</span>, x.shape)
|
||||
<span style="color: #228B22"># Decide degree on polynomial to fit</span>
|
||||
poly = PolynomialFeatures(degree = <span style="color: #B452CD">10</span>)
|
||||
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">500</span>
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">5</span>, nlambdas)
|
||||
<span style="color: #228B22"># Initialize a KFold instance</span>
|
||||
k = <span style="color: #B452CD">5</span>
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = <span style="color: #B452CD">0</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> lmb <span style="color: #8B008B">in</span> lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring=<span style="color: #CD5555">'neg_mean_squared_error'</span>, cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += <span style="color: #B452CD">1</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = <span style="color: #CD5555">'cross_val_score'</span>)
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We see from this plot that the optimal MSE occurs for a value of \( \lambda\in [10,100] \).
|
||||
In order to nail down the best value of \( \lambda \), we could in turn narrow down the search area.
|
||||
|
||||
<p>
|
||||
An alternative is to use the so-called grid search functionality included with the library <b>Scikit-Learn</b>, as demonstrated for the same example here.
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="grid-search">Grid Search </h2>
|
||||
|
||||
<p>
|
||||
An alternative is to use the so-called grid search functionality
|
||||
included with the library <b>Scikit-Learn</b>, as demonstrated for the same
|
||||
example here.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> KFold
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> Ridge
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> cross_val_score
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> PolynomialFeatures
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
np.random.seed(<span style="color: #B452CD">3155</span>)
|
||||
<span style="color: #228B22"># Generate the data.</span>
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.linspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">3</span>, n).reshape(-<span style="color: #B452CD">1</span>, <span style="color: #B452CD">1</span>)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)+ np.random.normal(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">0.1</span>, x.shape)
|
||||
<span style="color: #228B22"># Decide degree on polynomial to fit</span>
|
||||
poly = PolynomialFeatures(degree = <span style="color: #B452CD">10</span>)
|
||||
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">500</span>
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">3</span>, <span style="color: #B452CD">5</span>, nlambdas)
|
||||
<span style="color: #228B22"># Initialize a KFold instance</span>
|
||||
k = <span style="color: #B452CD">5</span>
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = <span style="color: #B452CD">0</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> lmb <span style="color: #8B008B">in</span> lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring=<span style="color: #CD5555">'neg_mean_squared_error'</span>, cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += <span style="color: #B452CD">1</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = <span style="color: #CD5555">'cross_val_score'</span>)
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
|
||||
@@ -72,6 +72,7 @@ div { text-align: justify; text-justify: inter-word; }
|
||||
2,
|
||||
None,
|
||||
'searching-for-optimal-regularization-parameters-lambda'),
|
||||
('Grid Search', 2, None, 'grid-search'),
|
||||
('Optimization, the central part of any Machine Learning '
|
||||
'algortithm',
|
||||
2,
|
||||
@@ -272,15 +273,105 @@ For a good discussion on gradient methods, see Goodfellow et al section 4.3-4.5
|
||||
<h2 id="searching-for-optimal-regularization-parameters-lambda">Searching for Optimal Regularization Parameters \( \lambda \) </h2>
|
||||
|
||||
<p>
|
||||
In project 1 when using Ridge and Lasso regression, we end up
|
||||
In project 1, when using Ridge and Lasso regression, we end up
|
||||
searching for the optimal parameter \( \lambda \) which minimizes our
|
||||
selected scores (MSE or \( R2 \) values for example). The brute force
|
||||
approach, as discussed in the code here for Ridge regression consists
|
||||
approach, as discussed in the code here for Ridge regression, consists
|
||||
in evaluating the MSE as function of different \( \lambda \) values.
|
||||
Based on these calculations, one tries then to determine the value of the hyperparameter \( \lambda \)
|
||||
which results in optimal scores (for example the smallest MSE or an \( R2=1 \)).
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> KFold
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> Ridge
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> cross_val_score
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> PolynomialFeatures
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
||||
<span style="color: #408080; font-style: italic"># Generate the data.</span>
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">-3</span>, <span style="color: #666666">3</span>, n)<span style="color: #666666">.</span>reshape(<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)<span style="color: #666666">+</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(<span style="color: #666666">0</span>, <span style="color: #666666">0.1</span>, x<span style="color: #666666">.</span>shape)
|
||||
<span style="color: #408080; font-style: italic"># Decide degree on polynomial to fit</span>
|
||||
poly <span style="color: #666666">=</span> PolynomialFeatures(degree <span style="color: #666666">=</span> <span style="color: #666666">10</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">500</span>
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-3</span>, <span style="color: #666666">5</span>, nlambdas)
|
||||
<span style="color: #408080; font-style: italic"># Initialize a KFold instance</span>
|
||||
k <span style="color: #666666">=</span> <span style="color: #666666">5</span>
|
||||
kfold <span style="color: #666666">=</span> KFold(n_splits <span style="color: #666666">=</span> k)
|
||||
estimated_mse_sklearn <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
i <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> lmb <span style="color: #AA22FF; font-weight: bold">in</span> lambdas:
|
||||
ridge <span style="color: #666666">=</span> Ridge(alpha <span style="color: #666666">=</span> lmb)
|
||||
estimated_mse_folds <span style="color: #666666">=</span> cross_val_score(ridge, x, y, scoring<span style="color: #666666">=</span><span style="color: #BA2121">'neg_mean_squared_error'</span>, cv<span style="color: #666666">=</span>kfold)
|
||||
estimated_mse_sklearn[i] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(<span style="color: #666666">-</span>estimated_mse_folds)
|
||||
i <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), estimated_mse_sklearn, label <span style="color: #666666">=</span> <span style="color: #BA2121">'cross_val_score'</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We see from this plot that the optimal MSE occurs for a value of \( \lambda\in [10,100] \).
|
||||
In order to nail down the best value of \( \lambda \), we could in turn narrow down the search area.
|
||||
|
||||
<p>
|
||||
An alternative is to use the so-called grid search functionality included with the library <b>Scikit-Learn</b>, as demonstrated for the same example here.
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="grid-search">Grid Search </h2>
|
||||
|
||||
<p>
|
||||
An alternative is to use the so-called grid search functionality
|
||||
included with the library <b>Scikit-Learn</b>, as demonstrated for the same
|
||||
example here.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> KFold
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> Ridge
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> cross_val_score
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> PolynomialFeatures
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
||||
<span style="color: #408080; font-style: italic"># Generate the data.</span>
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">-3</span>, <span style="color: #666666">3</span>, n)<span style="color: #666666">.</span>reshape(<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)<span style="color: #666666">+</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(<span style="color: #666666">0</span>, <span style="color: #666666">0.1</span>, x<span style="color: #666666">.</span>shape)
|
||||
<span style="color: #408080; font-style: italic"># Decide degree on polynomial to fit</span>
|
||||
poly <span style="color: #666666">=</span> PolynomialFeatures(degree <span style="color: #666666">=</span> <span style="color: #666666">10</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">500</span>
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-3</span>, <span style="color: #666666">5</span>, nlambdas)
|
||||
<span style="color: #408080; font-style: italic"># Initialize a KFold instance</span>
|
||||
k <span style="color: #666666">=</span> <span style="color: #666666">5</span>
|
||||
kfold <span style="color: #666666">=</span> KFold(n_splits <span style="color: #666666">=</span> k)
|
||||
estimated_mse_sklearn <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
i <span style="color: #666666">=</span> <span style="color: #666666">0</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> lmb <span style="color: #AA22FF; font-weight: bold">in</span> lambdas:
|
||||
ridge <span style="color: #666666">=</span> Ridge(alpha <span style="color: #666666">=</span> lmb)
|
||||
estimated_mse_folds <span style="color: #666666">=</span> cross_val_score(ridge, x, y, scoring<span style="color: #666666">=</span><span style="color: #BA2121">'neg_mean_squared_error'</span>, cv<span style="color: #666666">=</span>kfold)
|
||||
estimated_mse_sklearn[i] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(<span style="color: #666666">-</span>estimated_mse_folds)
|
||||
i <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), estimated_mse_sklearn, label <span style="color: #666666">=</span> <span style="color: #BA2121">'cross_val_score'</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
|
||||
Binary file not shown.
@@ -34,19 +34,128 @@
|
||||
"\n",
|
||||
"## Searching for Optimal Regularization Parameters $\\lambda$\n",
|
||||
"\n",
|
||||
"In project 1 when using Ridge and Lasso regression, we end up\n",
|
||||
"In project 1, when using Ridge and Lasso regression, we end up\n",
|
||||
"searching for the optimal parameter $\\lambda$ which minimizes our\n",
|
||||
"selected scores (MSE or $R2$ values for example). The brute force\n",
|
||||
"approach, as discussed in the code here for Ridge regression consists\n",
|
||||
"approach, as discussed in the code here for Ridge regression, consists\n",
|
||||
"in evaluating the MSE as function of different $\\lambda$ values.\n",
|
||||
"Based on these calculations, one tries then to determine the value of the hyperparameter $\\lambda$\n",
|
||||
"which results in optimal scores (for example the smallest MSE or an $R2=1$)."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": false,
|
||||
"editable": true
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%matplotlib inline\n",
|
||||
"\n",
|
||||
"An alternative is to use the so-called grid search functionality included with the library **Scikit-Learn**, as demonstrated for the same example here.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"import numpy as np\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"from sklearn.model_selection import KFold\n",
|
||||
"from sklearn.linear_model import Ridge\n",
|
||||
"from sklearn.model_selection import cross_val_score\n",
|
||||
"from sklearn.preprocessing import PolynomialFeatures\n",
|
||||
"\n",
|
||||
"# A seed just to ensure that the random numbers are the same for every run.\n",
|
||||
"np.random.seed(3155)\n",
|
||||
"# Generate the data.\n",
|
||||
"n = 100\n",
|
||||
"x = np.linspace(-3, 3, n).reshape(-1, 1)\n",
|
||||
"y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)\n",
|
||||
"# Decide degree on polynomial to fit\n",
|
||||
"poly = PolynomialFeatures(degree = 10)\n",
|
||||
"\n",
|
||||
"# Decide which values of lambda to use\n",
|
||||
"nlambdas = 500\n",
|
||||
"lambdas = np.logspace(-3, 5, nlambdas)\n",
|
||||
"# Initialize a KFold instance\n",
|
||||
"k = 5\n",
|
||||
"kfold = KFold(n_splits = k)\n",
|
||||
"estimated_mse_sklearn = np.zeros(nlambdas)\n",
|
||||
"i = 0\n",
|
||||
"for lmb in lambdas:\n",
|
||||
" ridge = Ridge(alpha = lmb)\n",
|
||||
" estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)\n",
|
||||
" estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)\n",
|
||||
" i += 1\n",
|
||||
"plt.figure()\n",
|
||||
"plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')\n",
|
||||
"plt.xlabel('log10(lambda)')\n",
|
||||
"plt.ylabel('MSE')\n",
|
||||
"plt.legend()\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"We see from this plot that the optimal MSE occurs for a value of $\\lambda\\in [10,100]$.\n",
|
||||
"In order to nail down the best value of $\\lambda$, we could in turn narrow down the search area.\n",
|
||||
"\n",
|
||||
"## Grid Search\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"An alternative is to use the so-called grid search functionality\n",
|
||||
"included with the library **Scikit-Learn**, as demonstrated for the same\n",
|
||||
"example here."
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": false,
|
||||
"editable": true
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import numpy as np\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"from sklearn.model_selection import KFold\n",
|
||||
"from sklearn.linear_model import Ridge\n",
|
||||
"from sklearn.model_selection import cross_val_score\n",
|
||||
"from sklearn.preprocessing import PolynomialFeatures\n",
|
||||
"\n",
|
||||
"# A seed just to ensure that the random numbers are the same for every run.\n",
|
||||
"np.random.seed(3155)\n",
|
||||
"# Generate the data.\n",
|
||||
"n = 100\n",
|
||||
"x = np.linspace(-3, 3, n).reshape(-1, 1)\n",
|
||||
"y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)\n",
|
||||
"# Decide degree on polynomial to fit\n",
|
||||
"poly = PolynomialFeatures(degree = 10)\n",
|
||||
"\n",
|
||||
"# Decide which values of lambda to use\n",
|
||||
"nlambdas = 500\n",
|
||||
"lambdas = np.logspace(-3, 5, nlambdas)\n",
|
||||
"# Initialize a KFold instance\n",
|
||||
"k = 5\n",
|
||||
"kfold = KFold(n_splits = k)\n",
|
||||
"estimated_mse_sklearn = np.zeros(nlambdas)\n",
|
||||
"i = 0\n",
|
||||
"for lmb in lambdas:\n",
|
||||
" ridge = Ridge(alpha = lmb)\n",
|
||||
" estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)\n",
|
||||
" estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)\n",
|
||||
" i += 1\n",
|
||||
"plt.figure()\n",
|
||||
"plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')\n",
|
||||
"plt.xlabel('log10(lambda)')\n",
|
||||
"plt.ylabel('MSE')\n",
|
||||
"plt.legend()\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## Optimization, the central part of any Machine Learning algortithm\n",
|
||||
"\n",
|
||||
"Almost every problem in machine learning and data science starts with\n",
|
||||
@@ -810,8 +919,6 @@
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"%matplotlib inline\n",
|
||||
"\n",
|
||||
"import numpy as np\n",
|
||||
"import numpy.linalg as la\n",
|
||||
"\n",
|
||||
|
||||
@@ -23,13 +23,99 @@ For a good discussion on gradient methods, see Goodfellow et al section 4.3-4.5
|
||||
!split
|
||||
===== Searching for Optimal Regularization Parameters $\lambda$ =====
|
||||
|
||||
In project 1 when using Ridge and Lasso regression, we end up
|
||||
In project 1, when using Ridge and Lasso regression, we end up
|
||||
searching for the optimal parameter $\lambda$ which minimizes our
|
||||
selected scores (MSE or $R2$ values for example). The brute force
|
||||
approach, as discussed in the code here for Ridge regression consists
|
||||
approach, as discussed in the code here for Ridge regression, consists
|
||||
in evaluating the MSE as function of different $\lambda$ values.
|
||||
Based on these calculations, one tries then to determine the value of the hyperparameter $\lambda$
|
||||
which results in optimal scores (for example the smallest MSE or an $R2=1$).
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
An alternative is to use the so-called grid search functionality included with the library _Scikit-Learn_, as demonstrated for the same example here.
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
We see from this plot that the optimal MSE occurs for a value of $\lambda\in [10,100]$.
|
||||
In order to nail down the best value of $\lambda$, we could in turn narrow down the search area.
|
||||
|
||||
!split
|
||||
===== Grid Search =====
|
||||
|
||||
|
||||
An alternative is to use the so-called grid search functionality
|
||||
included with the library _Scikit-Learn_, as demonstrated for the same
|
||||
example here.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import KFold
|
||||
from sklearn.linear_model import Ridge
|
||||
from sklearn.model_selection import cross_val_score
|
||||
from sklearn.preprocessing import PolynomialFeatures
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
np.random.seed(3155)
|
||||
# Generate the data.
|
||||
n = 100
|
||||
x = np.linspace(-3, 3, n).reshape(-1, 1)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)+ np.random.normal(0, 0.1, x.shape)
|
||||
# Decide degree on polynomial to fit
|
||||
poly = PolynomialFeatures(degree = 10)
|
||||
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 500
|
||||
lambdas = np.logspace(-3, 5, nlambdas)
|
||||
# Initialize a KFold instance
|
||||
k = 5
|
||||
kfold = KFold(n_splits = k)
|
||||
estimated_mse_sklearn = np.zeros(nlambdas)
|
||||
i = 0
|
||||
for lmb in lambdas:
|
||||
ridge = Ridge(alpha = lmb)
|
||||
estimated_mse_folds = cross_val_score(ridge, x, y, scoring='neg_mean_squared_error', cv=kfold)
|
||||
estimated_mse_sklearn[i] = np.mean(-estimated_mse_folds)
|
||||
i += 1
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), estimated_mse_sklearn, label = 'cross_val_score')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
@@ -1290,3 +1376,5 @@ plt.show()
|
||||
_Challenge_: try to write a similar code for a Logistic Regression case.
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user