update week38
This commit is contained in:
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
@@ -234,11 +237,153 @@ MathJax.Hub.Config({
|
||||
<a name="part0007"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="to-think-about" class="anchor">To think about </h2>
|
||||
<h2 id="to-think-about-first-part" class="anchor">To think about, first part </h2>
|
||||
|
||||
<p>
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s library, there are some minor things to keep in mind.
|
||||
The example here shows how one can keep the intercept in order to compare own code.
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some things to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
|
||||
<p>
|
||||
The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which may confuse us.
|
||||
|
||||
<p>
|
||||
Yes, it could be a bad idea to include the intercept column for the
|
||||
exact reason you stated. If no transformation is applied to your data,
|
||||
the intercept can be interpreted as the expected value of your target
|
||||
variable when all your predictors are put to zero. Therefore, whenever
|
||||
you cannot assume that the expected target variable is zero when all
|
||||
your predictors are zero, it could be a bad idea to apply a model
|
||||
which penalizes the intercept. Also, the analytical solution to the
|
||||
ridge regression coefficients (when not shrinking $$\beta_0$$) is
|
||||
derived under the assumption that both y and X are zero centered (mean
|
||||
subtracted). What you are doing is correct, but you should also zero
|
||||
center X (subtracting the mean of each column from the corresponding
|
||||
column). 
|
||||
|
||||
<p>
|
||||
If your predictors are of different scales, I would advice you to
|
||||
standardize X by subtracting the mean of each column from the
|
||||
corresponding column and dividing the column with its standard
|
||||
deviation. If you dont do this, you will give an "unfair" penalization
|
||||
of the parameters since their magnitude depends on the scale of their
|
||||
corresponding predictor. Suppose that you have an input variable
|
||||
"height". Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters. You may see how
|
||||
this could become a problem when considering the loss function for
|
||||
ridge regression.
|
||||
|
||||
<p>
|
||||
Remember that when you do any transformation to your dataset before
|
||||
training, the exact same transformation has to be applied to new data
|
||||
before making a prediction. In your case, this means:
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #408080; font-style: italic">#Model training:</span>
|
||||
y_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
X_train <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
||||
y_train <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_train_mean
|
||||
|
||||
trained_model <span style="color: #666666">=</span> some_model<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
|
||||
<span style="color: #408080; font-style: italic">#Model prediction:</span>
|
||||
X_test <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean <span style="color: #408080; font-style: italic">#Use mean from training data</span>
|
||||
y_pred <span style="color: #666666">=</span> trained_model(X_test)
|
||||
y_pred <span style="color: #666666">=</span> y_pred <span style="color: #666666">+</span> y_train_mean
|
||||
</pre></div>
|
||||
<p>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
which gives
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
$$
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> X <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(X,axis<span style="color: #666666">=0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
can be rewritten as
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
$$
|
||||
|
||||
or
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
|
||||
<p>
|
||||
which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
|
||||
@@ -327,6 +472,183 @@ plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</s
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">R2</span>(y_data, y_model):
|
||||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span> <span style="color: #666666">-</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> y_model) <span style="color: #666666">**</span> <span style="color: #666666">2</span>) <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(y_data)) <span style="color: #666666">**</span> <span style="color: #666666">2</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">5</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean <span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train) <span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler <span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">1</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> intercept_ <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #408080; font-style: italic">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #008000">print</span>(intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">2021</span>)
|
||||
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">fit_beta</span>(X, y):
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X) <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
||||
|
||||
|
||||
true_beta <span style="color: #666666">=</span> [<span style="color: #666666">2</span>, <span style="color: #666666">0.5</span>, <span style="color: #666666">3.7</span>]
|
||||
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>, <span style="color: #666666">1</span>, <span style="color: #666666">11</span>)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(
|
||||
np<span style="color: #666666">.</span>asarray([x <span style="color: #666666">**</span> p <span style="color: #666666">*</span> b <span style="color: #008000; font-weight: bold">for</span> p, b <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(true_beta)]), axis<span style="color: #666666">=0</span>
|
||||
) <span style="color: #666666">+</span> <span style="color: #666666">0.1</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(size<span style="color: #666666">=</span><span style="color: #008000">len</span>(x))
|
||||
|
||||
degree <span style="color: #666666">=</span> <span style="color: #666666">3</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree))
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> p
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is included in the design matrix</span>
|
||||
clf <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"True beta: </span><span style="color: #BB6688; font-weight: bold">{</span>true_beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>clf<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>scatter(x, y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Data"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, clf<span style="color: #666666">.</span>predict(X), label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Do not include the intercept in the design matrix</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> (p <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is not included in the design matrix</span>
|
||||
clf <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(y, axis<span style="color: #666666">=0</span>)
|
||||
X_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(X, axis<span style="color: #666666">=0</span>)
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X <span style="color: #666666">-</span> X_offset, y <span style="color: #666666">-</span> y_offset)
|
||||
intercept <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_offset <span style="color: #666666">-</span> X_offset <span style="color: #666666">@</span> beta)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Manual intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>intercept<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta (sans intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>clf<span style="color: #666666">.</span>intercept_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta (sans intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>clf<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta <span style="color: #666666">+</span> intercept, <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit (manual intercept)"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, clf<span style="color: #666666">.</span>predict(X), <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt<span style="color: #666666">.</span>grid()
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -63,7 +63,10 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -187,7 +190,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs004.html#how-to-set-up-the-cross-validation-for-ridge-and-or-lasso" style="font-size: 80%;">How to set up the cross-validation for Ridge and/or Lasso</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs005.html#cross-validation-in-brief" style="font-size: 80%;">Cross-validation in brief</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs006.html#code-example-for-cross-validation-and-k-fold-cross-validation" style="font-size: 80%;">Code Example for Cross-validation and \( k \)-fold Cross-validation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about" style="font-size: 80%;">To think about</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs007.html#to-think-about-first-part" style="font-size: 80%;">To think about, first part</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs008.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
|
||||
@@ -419,11 +419,175 @@ plt.show()
|
||||
|
||||
|
||||
<section>
|
||||
<h2 id="to-think-about">To think about </h2>
|
||||
<h2 id="to-think-about-first-part">To think about, first part </h2>
|
||||
|
||||
<p>
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s library, there are some minor things to keep in mind.
|
||||
The example here shows how one can keep the intercept in order to compare own code.
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some things to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
|
||||
<p>
|
||||
The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which may confuse us.
|
||||
|
||||
<p>
|
||||
Yes, it could be a bad idea to include the intercept column for the
|
||||
exact reason you stated. If no transformation is applied to your data,
|
||||
the intercept can be interpreted as the expected value of your target
|
||||
variable when all your predictors are put to zero. Therefore, whenever
|
||||
you cannot assume that the expected target variable is zero when all
|
||||
your predictors are zero, it could be a bad idea to apply a model
|
||||
which penalizes the intercept. Also, the analytical solution to the
|
||||
ridge regression coefficients (when not shrinking <p> <br>
|
||||
$$\beta_0$$
|
||||
<p> <br>) is
|
||||
derived under the assumption that both y and X are zero centered (mean
|
||||
subtracted). What you are doing is correct, but you should also zero
|
||||
center X (subtracting the mean of each column from the corresponding
|
||||
column). 
|
||||
|
||||
<p>
|
||||
If your predictors are of different scales, I would advice you to
|
||||
standardize X by subtracting the mean of each column from the
|
||||
corresponding column and dividing the column with its standard
|
||||
deviation. If you dont do this, you will give an "unfair" penalization
|
||||
of the parameters since their magnitude depends on the scale of their
|
||||
corresponding predictor. Suppose that you have an input variable
|
||||
"height". Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters. You may see how
|
||||
this could become a problem when considering the loss function for
|
||||
ridge regression.
|
||||
|
||||
<p>
|
||||
Remember that when you do any transformation to your dataset before
|
||||
training, the exact same transformation has to be applied to new data
|
||||
before making a prediction. In your case, this means:
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span><span style="color: #228B22">#Model training:</span>
|
||||
y_train_mean = np.mean(y_train)
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
X_train = X_train - X_train_mean
|
||||
y_train = y_train - y_train_mean
|
||||
|
||||
trained_model = some_model.fit(X_train,y_train)
|
||||
|
||||
<span style="color: #228B22">#Model prediction:</span>
|
||||
X_test = X_test - X_train_mean <span style="color: #228B22">#Use mean from training data</span>
|
||||
y_pred = trained_model(X_test)
|
||||
y_pred = y_pred + y_train_mean
|
||||
</pre></div>
|
||||
<p>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
<p> <br>
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
which gives
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span>X = X - np.mean(X,axis=<span style="color: #B452CD">0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
can be rewritten as
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
resulting in
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
<p> <br>
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
or
|
||||
<p> <br>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
<p> <br>
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
|
||||
@@ -512,6 +676,183 @@ plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">pandas</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pd</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">R2</span>(y_data, y_model):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span> - np.sum((y_data - y_model) ** <span style="color: #B452CD">2</span>) / np.sum((y_data - np.mean(y_data)) ** <span style="color: #B452CD">2</span>)
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">315</span>)
|
||||
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">5</span>
|
||||
X = np.zeros((n,Maxpolydegree-<span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,Maxpolydegree): <span style="color: #228B22">#No intercept column</span>
|
||||
X[:,degree-<span style="color: #B452CD">1</span>] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
X_train_scaled = X_train - X_train_mean <span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) <span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled = y_train - y_scaler <span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
|
||||
|
||||
p = Maxpolydegree-<span style="color: #B452CD">1</span>
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">4</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">1</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean<span style="color: #707a7c">@OwnRidgeBeta</span> <span style="color: #228B22">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #228B22">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta) <span style="color: #228B22">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #658b00">print</span>(intercept_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'b--'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g--'</span>, label = <span style="color: #CD5555">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np.random.seed(<span style="color: #B452CD">2021</span>)
|
||||
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">fit_beta</span>(X, y):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.linalg.pinv(X.T @ X) @ X.T @ y
|
||||
|
||||
|
||||
true_beta = [<span style="color: #B452CD">2</span>, <span style="color: #B452CD">0.5</span>, <span style="color: #B452CD">3.7</span>]
|
||||
|
||||
x = np.linspace(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">1</span>, <span style="color: #B452CD">11</span>)
|
||||
y = np.sum(
|
||||
np.asarray([x ** p * b <span style="color: #8B008B; font-weight: bold">for</span> p, b <span style="color: #8B008B">in</span> <span style="color: #658b00">enumerate</span>(true_beta)]), axis=<span style="color: #B452CD">0</span>
|
||||
) + <span style="color: #B452CD">0.1</span> * np.random.normal(size=<span style="color: #658b00">len</span>(x))
|
||||
|
||||
degree = <span style="color: #B452CD">3</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree))
|
||||
|
||||
<span style="color: #228B22"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree):
|
||||
X[:, p] = x ** p
|
||||
|
||||
beta = fit_beta(X, y)
|
||||
|
||||
<span style="color: #228B22"># Intercept is included in the design matrix</span>
|
||||
clf = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>).fit(X, y)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"True beta: {</span>true_beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta: {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta: {</span>clf.coef_<span style="color: #CD5555">}"</span>)
|
||||
|
||||
|
||||
plt.figure()
|
||||
plt.scatter(x, y, label=<span style="color: #CD5555">"Data"</span>)
|
||||
plt.plot(x, X @ beta, label=<span style="color: #CD5555">"Fit"</span>)
|
||||
plt.plot(x, clf.predict(X), label=<span style="color: #CD5555">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #228B22"># Do not include the intercept in the design matrix</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree - <span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree - <span style="color: #B452CD">1</span>):
|
||||
X[:, p] = x ** (p + <span style="color: #B452CD">1</span>)
|
||||
|
||||
<span style="color: #228B22"># Intercept is not included in the design matrix</span>
|
||||
clf = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">True</span>).fit(X, y)
|
||||
|
||||
<span style="color: #228B22"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset = np.average(y, axis=<span style="color: #B452CD">0</span>)
|
||||
X_offset = np.average(X, axis=<span style="color: #B452CD">0</span>)
|
||||
|
||||
beta = fit_beta(X - X_offset, y - y_offset)
|
||||
intercept = np.mean(y_offset - X_offset @ beta)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Manual intercept: {</span>intercept<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta (sans intercept): {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn intercept: {</span>clf.intercept_<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta (sans intercept): {</span>clf.coef_<span style="color: #CD5555">}"</span>)
|
||||
|
||||
plt.plot(x, X @ beta + intercept, <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Fit (manual intercept)"</span>)
|
||||
plt.plot(x, clf.predict(X), <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt.grid()
|
||||
plt.legend()
|
||||
|
||||
plt.show()
|
||||
</pre></div>
|
||||
</section>
|
||||
|
||||
@@ -57,7 +57,10 @@ div { text-align: justify; text-justify: inter-word; }
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -426,11 +429,153 @@ plt.show()
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="to-think-about">To think about </h2>
|
||||
<h2 id="to-think-about-first-part">To think about, first part </h2>
|
||||
|
||||
<p>
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s library, there are some minor things to keep in mind.
|
||||
The example here shows how one can keep the intercept in order to compare own code.
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some things to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
|
||||
<p>
|
||||
The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which may confuse us.
|
||||
|
||||
<p>
|
||||
Yes, it could be a bad idea to include the intercept column for the
|
||||
exact reason you stated. If no transformation is applied to your data,
|
||||
the intercept can be interpreted as the expected value of your target
|
||||
variable when all your predictors are put to zero. Therefore, whenever
|
||||
you cannot assume that the expected target variable is zero when all
|
||||
your predictors are zero, it could be a bad idea to apply a model
|
||||
which penalizes the intercept. Also, the analytical solution to the
|
||||
ridge regression coefficients (when not shrinking $$\beta_0$$) is
|
||||
derived under the assumption that both y and X are zero centered (mean
|
||||
subtracted). What you are doing is correct, but you should also zero
|
||||
center X (subtracting the mean of each column from the corresponding
|
||||
column). 
|
||||
|
||||
<p>
|
||||
If your predictors are of different scales, I would advice you to
|
||||
standardize X by subtracting the mean of each column from the
|
||||
corresponding column and dividing the column with its standard
|
||||
deviation. If you dont do this, you will give an "unfair" penalization
|
||||
of the parameters since their magnitude depends on the scale of their
|
||||
corresponding predictor. Suppose that you have an input variable
|
||||
"height". Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters. You may see how
|
||||
this could become a problem when considering the loss function for
|
||||
ridge regression.
|
||||
|
||||
<p>
|
||||
Remember that when you do any transformation to your dataset before
|
||||
training, the exact same transformation has to be applied to new data
|
||||
before making a prediction. In your case, this means:
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span><span style="color: #228B22">#Model training:</span>
|
||||
y_train_mean = np.mean(y_train)
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
X_train = X_train - X_train_mean
|
||||
y_train = y_train - y_train_mean
|
||||
|
||||
trained_model = some_model.fit(X_train,y_train)
|
||||
|
||||
<span style="color: #228B22">#Model prediction:</span>
|
||||
X_test = X_test - X_train_mean <span style="color: #228B22">#Use mean from training data</span>
|
||||
y_pred = trained_model(X_test)
|
||||
y_pred = y_pred + y_train_mean
|
||||
</pre></div>
|
||||
<p>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
which gives
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
$$
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span>X = X - np.mean(X,axis=<span style="color: #B452CD">0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
can be rewritten as
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
$$
|
||||
|
||||
or
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
|
||||
<p>
|
||||
which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
|
||||
@@ -519,6 +664,183 @@ plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">pandas</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pd</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">R2</span>(y_data, y_model):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span> - np.sum((y_data - y_model) ** <span style="color: #B452CD">2</span>) / np.sum((y_data - np.mean(y_data)) ** <span style="color: #B452CD">2</span>)
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">315</span>)
|
||||
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">5</span>
|
||||
X = np.zeros((n,Maxpolydegree-<span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,Maxpolydegree): <span style="color: #228B22">#No intercept column</span>
|
||||
X[:,degree-<span style="color: #B452CD">1</span>] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
X_train_scaled = X_train - X_train_mean <span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) <span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled = y_train - y_scaler <span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
|
||||
|
||||
p = Maxpolydegree-<span style="color: #B452CD">1</span>
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">4</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">1</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean<span style="color: #707a7c">@OwnRidgeBeta</span> <span style="color: #228B22">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #228B22">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta) <span style="color: #228B22">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #658b00">print</span>(intercept_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'b--'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g--'</span>, label = <span style="color: #CD5555">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np.random.seed(<span style="color: #B452CD">2021</span>)
|
||||
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">fit_beta</span>(X, y):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.linalg.pinv(X.T @ X) @ X.T @ y
|
||||
|
||||
|
||||
true_beta = [<span style="color: #B452CD">2</span>, <span style="color: #B452CD">0.5</span>, <span style="color: #B452CD">3.7</span>]
|
||||
|
||||
x = np.linspace(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">1</span>, <span style="color: #B452CD">11</span>)
|
||||
y = np.sum(
|
||||
np.asarray([x ** p * b <span style="color: #8B008B; font-weight: bold">for</span> p, b <span style="color: #8B008B">in</span> <span style="color: #658b00">enumerate</span>(true_beta)]), axis=<span style="color: #B452CD">0</span>
|
||||
) + <span style="color: #B452CD">0.1</span> * np.random.normal(size=<span style="color: #658b00">len</span>(x))
|
||||
|
||||
degree = <span style="color: #B452CD">3</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree))
|
||||
|
||||
<span style="color: #228B22"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree):
|
||||
X[:, p] = x ** p
|
||||
|
||||
beta = fit_beta(X, y)
|
||||
|
||||
<span style="color: #228B22"># Intercept is included in the design matrix</span>
|
||||
clf = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>).fit(X, y)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"True beta: {</span>true_beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta: {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta: {</span>clf.coef_<span style="color: #CD5555">}"</span>)
|
||||
|
||||
|
||||
plt.figure()
|
||||
plt.scatter(x, y, label=<span style="color: #CD5555">"Data"</span>)
|
||||
plt.plot(x, X @ beta, label=<span style="color: #CD5555">"Fit"</span>)
|
||||
plt.plot(x, clf.predict(X), label=<span style="color: #CD5555">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #228B22"># Do not include the intercept in the design matrix</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree - <span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree - <span style="color: #B452CD">1</span>):
|
||||
X[:, p] = x ** (p + <span style="color: #B452CD">1</span>)
|
||||
|
||||
<span style="color: #228B22"># Intercept is not included in the design matrix</span>
|
||||
clf = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">True</span>).fit(X, y)
|
||||
|
||||
<span style="color: #228B22"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset = np.average(y, axis=<span style="color: #B452CD">0</span>)
|
||||
X_offset = np.average(X, axis=<span style="color: #B452CD">0</span>)
|
||||
|
||||
beta = fit_beta(X - X_offset, y - y_offset)
|
||||
intercept = np.mean(y_offset - X_offset @ beta)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Manual intercept: {</span>intercept<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta (sans intercept): {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn intercept: {</span>clf.intercept_<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta (sans intercept): {</span>clf.coef_<span style="color: #CD5555">}"</span>)
|
||||
|
||||
plt.plot(x, X @ beta + intercept, <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Fit (manual intercept)"</span>)
|
||||
plt.plot(x, clf.predict(X), <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt.grid()
|
||||
plt.legend()
|
||||
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
@@ -62,7 +62,10 @@ div { text-align: justify; text-justify: inter-word; }
|
||||
2,
|
||||
None,
|
||||
'code-example-for-cross-validation-and-k-fold-cross-validation'),
|
||||
('To think about', 2, None, 'to-think-about'),
|
||||
('To think about, first part',
|
||||
2,
|
||||
None,
|
||||
'to-think-about-first-part'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -431,11 +434,153 @@ plt<span style="color: #666666">.</span>show()
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="to-think-about">To think about </h2>
|
||||
<h2 id="to-think-about-first-part">To think about, first part </h2>
|
||||
|
||||
<p>
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s library, there are some minor things to keep in mind.
|
||||
The example here shows how one can keep the intercept in order to compare own code.
|
||||
When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some things to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
|
||||
<p>
|
||||
The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which may confuse us.
|
||||
|
||||
<p>
|
||||
Yes, it could be a bad idea to include the intercept column for the
|
||||
exact reason you stated. If no transformation is applied to your data,
|
||||
the intercept can be interpreted as the expected value of your target
|
||||
variable when all your predictors are put to zero. Therefore, whenever
|
||||
you cannot assume that the expected target variable is zero when all
|
||||
your predictors are zero, it could be a bad idea to apply a model
|
||||
which penalizes the intercept. Also, the analytical solution to the
|
||||
ridge regression coefficients (when not shrinking $$\beta_0$$) is
|
||||
derived under the assumption that both y and X are zero centered (mean
|
||||
subtracted). What you are doing is correct, but you should also zero
|
||||
center X (subtracting the mean of each column from the corresponding
|
||||
column). 
|
||||
|
||||
<p>
|
||||
If your predictors are of different scales, I would advice you to
|
||||
standardize X by subtracting the mean of each column from the
|
||||
corresponding column and dividing the column with its standard
|
||||
deviation. If you dont do this, you will give an "unfair" penalization
|
||||
of the parameters since their magnitude depends on the scale of their
|
||||
corresponding predictor. Suppose that you have an input variable
|
||||
"height". Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters. You may see how
|
||||
this could become a problem when considering the loss function for
|
||||
ridge regression.
|
||||
|
||||
<p>
|
||||
Remember that when you do any transformation to your dataset before
|
||||
training, the exact same transformation has to be applied to new data
|
||||
before making a prediction. In your case, this means:
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #408080; font-style: italic">#Model training:</span>
|
||||
y_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
X_train <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
||||
y_train <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_train_mean
|
||||
|
||||
trained_model <span style="color: #666666">=</span> some_model<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
|
||||
<span style="color: #408080; font-style: italic">#Model prediction:</span>
|
||||
X_test <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean <span style="color: #408080; font-style: italic">#Use mean from training data</span>
|
||||
y_pred <span style="color: #666666">=</span> trained_model(X_test)
|
||||
y_pred <span style="color: #666666">=</span> y_pred <span style="color: #666666">+</span> y_train_mean
|
||||
</pre></div>
|
||||
<p>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
which gives
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
$$
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> X <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(X,axis<span style="color: #666666">=0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
can be rewritten as
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
$$
|
||||
|
||||
or
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
|
||||
<p>
|
||||
which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
|
||||
@@ -524,6 +669,183 @@ plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</s
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">R2</span>(y_data, y_model):
|
||||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span> <span style="color: #666666">-</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> y_model) <span style="color: #666666">**</span> <span style="color: #666666">2</span>) <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(y_data)) <span style="color: #666666">**</span> <span style="color: #666666">2</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">5</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean <span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train) <span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler <span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">1</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> intercept_ <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #408080; font-style: italic">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #008000">print</span>(intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">2021</span>)
|
||||
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">fit_beta</span>(X, y):
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X) <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
||||
|
||||
|
||||
true_beta <span style="color: #666666">=</span> [<span style="color: #666666">2</span>, <span style="color: #666666">0.5</span>, <span style="color: #666666">3.7</span>]
|
||||
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>, <span style="color: #666666">1</span>, <span style="color: #666666">11</span>)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(
|
||||
np<span style="color: #666666">.</span>asarray([x <span style="color: #666666">**</span> p <span style="color: #666666">*</span> b <span style="color: #008000; font-weight: bold">for</span> p, b <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(true_beta)]), axis<span style="color: #666666">=0</span>
|
||||
) <span style="color: #666666">+</span> <span style="color: #666666">0.1</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(size<span style="color: #666666">=</span><span style="color: #008000">len</span>(x))
|
||||
|
||||
degree <span style="color: #666666">=</span> <span style="color: #666666">3</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree))
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> p
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is included in the design matrix</span>
|
||||
clf <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"True beta: </span><span style="color: #BB6688; font-weight: bold">{</span>true_beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>clf<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>scatter(x, y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Data"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, clf<span style="color: #666666">.</span>predict(X), label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Do not include the intercept in the design matrix</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> (p <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is not included in the design matrix</span>
|
||||
clf <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(y, axis<span style="color: #666666">=0</span>)
|
||||
X_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(X, axis<span style="color: #666666">=0</span>)
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X <span style="color: #666666">-</span> X_offset, y <span style="color: #666666">-</span> y_offset)
|
||||
intercept <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_offset <span style="color: #666666">-</span> X_offset <span style="color: #666666">@</span> beta)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Manual intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>intercept<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta (sans intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>clf<span style="color: #666666">.</span>intercept_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta (sans intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>clf<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta <span style="color: #666666">+</span> intercept, <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit (manual intercept)"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, clf<span style="color: #666666">.</span>predict(X), <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt<span style="color: #666666">.</span>grid()
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
Binary file not shown.
@@ -17,6 +17,9 @@
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"## Plans for week 38\n",
|
||||
"\n",
|
||||
"* Thursday: Summary of regression methods and discussion of project 1. Start Logistic Regression\n",
|
||||
@@ -347,10 +350,277 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## To think about\n",
|
||||
"## To think about, first part\n",
|
||||
"\n",
|
||||
"When you are comparing your own code with for example **Scikit-Learn**'s library, there are some minor things to keep in mind.\n",
|
||||
"The example here shows how one can keep the intercept in order to compare own code."
|
||||
"When you are comparing your own code with for example **Scikit-Learn**'s\n",
|
||||
"library, there are some things to keep in mind. The examples\n",
|
||||
"here demonstrate some of these aspects with potential pitfalls.\n",
|
||||
"\n",
|
||||
"The discussion here focuses on the role of the intercept, how we can\n",
|
||||
"set up the design matrix, what scaling we should use and other topics\n",
|
||||
"which may confuse us.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Yes, it could be a bad idea to include the intercept column for the\n",
|
||||
"exact reason you stated. If no transformation is applied to your data,\n",
|
||||
"the intercept can be interpreted as the expected value of your target\n",
|
||||
"variable when all your predictors are put to zero. Therefore, whenever\n",
|
||||
"you cannot assume that the expected target variable is zero when all\n",
|
||||
"your predictors are zero, it could be a bad idea to apply a model\n",
|
||||
"which penalizes the intercept. Also, the analytical solution to the\n",
|
||||
"ridge regression coefficients (when not shrinking $$\\beta_0$$) is\n",
|
||||
"derived under the assumption that both y and X are zero centered (mean\n",
|
||||
"subtracted). What you are doing is correct, but you should also zero\n",
|
||||
"center X (subtracting the mean of each column from the corresponding\n",
|
||||
"column). \n",
|
||||
"\n",
|
||||
"If your predictors are of different scales, I would advice you to\n",
|
||||
"standardize X by subtracting the mean of each column from the\n",
|
||||
"corresponding column and dividing the column with its standard\n",
|
||||
"deviation. If you dont do this, you will give an \"unfair\" penalization\n",
|
||||
"of the parameters since their magnitude depends on the scale of their\n",
|
||||
"corresponding predictor. Suppose that you have an input variable\n",
|
||||
"\"height\". Human height might be measured in inches or meters or\n",
|
||||
"kilometers. If measured in kilometers, a standard linear regression\n",
|
||||
"model with this predictor would probably give a much bigger\n",
|
||||
"coefficient term, than if measured in millimeters. You may see how\n",
|
||||
"this could become a problem when considering the loss function for\n",
|
||||
"ridge regression.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Remember that when you do any transformation to your dataset before\n",
|
||||
"training, the exact same transformation has to be applied to new data\n",
|
||||
"before making a prediction. In your case, this means:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": false,
|
||||
"editable": true
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"#Model training:\n",
|
||||
"y_train_mean = np.mean(y_train)\n",
|
||||
"X_train_mean = np.mean(X_train,axis=0)\n",
|
||||
"X_train = X_train - X_train_mean\n",
|
||||
"y_train = y_train - y_train_mean\n",
|
||||
"\n",
|
||||
"trained_model = some_model.fit(X_train,y_train)\n",
|
||||
"\n",
|
||||
"#Model prediction:\n",
|
||||
"X_test = X_test - X_train_mean #Use mean from training data\n",
|
||||
"y_pred = trained_model(X_test)\n",
|
||||
"y_pred = y_pred + y_train_mean"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"Here is a mathematical explanation of the zero centering:\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"The cost/loss function for Ridge regression is:"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"C(\\beta_0, \\beta_1, ... , \\beta_P) = \\sum_{i=1}^{n} (y_i - \\beta_0 - \\sum_{p=1}^P X_{ip}\\beta_p)^2 + \\lambda \\sum_{p=1}^P \\beta_p^2.\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"Notice that the intercept is left out of the $L_2$ regularization term. The design matrix\n",
|
||||
"$X$ does in this case not contain any intercept column. We want"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\frac{\\partial L}{\\partial \\beta_j} = 0,\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"for all $j$, so lets start with $\\beta_0$. This means that we have"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\frac{\\partial L}{\\partial \\beta_0} = -2\\sum_{i=1}^{n} (y_i - \\beta_0 - \\sum_{p=1}^P X_{ip} \\beta_p).\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"We want to solve"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"-2\\sum_{i=1}^{n} (y_i - \\beta_0 - \\sum_{p=1}^P X_{ip} \\beta_p) = 0,\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"which gives"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\sum_{i=1}^{n} \\beta_0 = \\sum_{i=1}^{n}y_i - \\sum_{i=1}^{n} \\sum_{p=1}^P X_{ip} \\beta_p,\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"or\n",
|
||||
"$ n\\beta_0 = \\sum_{i=1}^{n} y_i - \\sum_{p=1}^P\\beta_p \\sum_{i=1}^{n} X_{ip}$.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"If we assume that every column of $X$ is centered, whic we can do by subtracting the mean,"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": false,
|
||||
"editable": true
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"X = X - np.mean(X,axis=0)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"the sum $ \\sum_{i=1}^{n} X_{ip} $\n",
|
||||
"\n",
|
||||
"can be rewritten as"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\sum_{i=1}^{n} (X_{ip} - \\frac{1}{n}\\sum_{i=1}^{n} X_{ip}) = \\sum_{i=1}^{n} X_{ip} - \\sum_{i=1}^{n} \\frac{1}{n} \\sum_{i=1}^{n}X_{ip},\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"resulting in"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\sum_{i=1}^{n} X_{ip} - n \\frac{1}{n} \\sum_{i=1}^{n}X_{ip} = 0.\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"Finally we have"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"n\\beta_0 = \\sum_{i=1}^{n} y_i - \\sum_{p=1}^P\\beta_p \\sum_{i=1}^{n} X_{ip},\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"or"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\beta_0 = \\frac{1}{n}\\sum_{i=1}^{n} y_i = y_{average}.\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"Replacing $y_i$ with $y_i - \\beta_0 = y_i - y_{average}$ in the loss function will give us (in vector-matrix disguise)"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"C(\\boldsymbol{\\beta}) = (\\boldsymbol{\\tilde{y}} - \\tilde{X}\\boldsymbol{\\beta})^T(\\boldsymbol{\\tilde{y}} - \\tilde{X}\\boldsymbol{\\beta}) + \\lambda \\boldsymbol{\\beta}^T\\boldsymbol{\\beta},\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"which has the solution\n",
|
||||
"\n",
|
||||
"$\\beta = (\\tilde{X}^T\\tilde{X} + \\lambda I)^{-1}\\tilde{X}^T\\boldsymbol{\\tilde{y}}$.\n",
|
||||
"where $\\boldsymbol{\\tilde{y}} = \\boldsymbol{y} - y_{average}$\n",
|
||||
"and $\\tilde{X}_{ij} = X_{ij} - \\frac{1}{n}\\sum_{k=1}^{n-1}X_{kj}$."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -449,6 +719,197 @@
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": false,
|
||||
"editable": true
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import numpy as np\n",
|
||||
"import pandas as pd\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"from sklearn.model_selection import train_test_split\n",
|
||||
"from sklearn import linear_model\n",
|
||||
"from sklearn.preprocessing import StandardScaler\n",
|
||||
"\n",
|
||||
"def R2(y_data, y_model):\n",
|
||||
" return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n",
|
||||
"def MSE(y_data,y_model):\n",
|
||||
" n = np.size(y_model)\n",
|
||||
" return np.sum((y_data-y_model)**2)/n\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# A seed just to ensure that the random numbers are the same for every run.\n",
|
||||
"# Useful for eventual debugging.\n",
|
||||
"np.random.seed(315)\n",
|
||||
"\n",
|
||||
"n = 100\n",
|
||||
"x = np.random.rand(n)\n",
|
||||
"y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n",
|
||||
"\n",
|
||||
"Maxpolydegree = 5\n",
|
||||
"X = np.zeros((n,Maxpolydegree-1))\n",
|
||||
"\n",
|
||||
"for degree in range(1,Maxpolydegree): #No intercept column\n",
|
||||
" X[:,degree-1] = x**(degree)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# We split the data in test and training data\n",
|
||||
"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable\n",
|
||||
"X_train_mean = np.mean(X_train,axis=0)\n",
|
||||
"X_train_scaled = X_train - X_train_mean #Center by removing mean from each feature\n",
|
||||
"X_test_scaled = X_test - X_train_mean\n",
|
||||
"\n",
|
||||
"y_scaler = np.mean(y_train) #The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)\n",
|
||||
"y_train_scaled = y_train - y_scaler #Remove the intercept from the training data.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"p = Maxpolydegree-1\n",
|
||||
"I = np.eye(p,p)\n",
|
||||
"# Decide which values of lambda to use\n",
|
||||
"nlambdas = 4\n",
|
||||
"MSEOwnRidgePredict = np.zeros(nlambdas)\n",
|
||||
"MSERidgePredict = np.zeros(nlambdas)\n",
|
||||
"\n",
|
||||
"lambdas = np.logspace(-4, 1, nlambdas)\n",
|
||||
"for i in range(nlambdas):\n",
|
||||
" lmb = lambdas[i]\n",
|
||||
" OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)\n",
|
||||
" intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data\n",
|
||||
" \n",
|
||||
" ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ #Add intercept to prediction\n",
|
||||
" #EQUIVALENT PREDICTION:\n",
|
||||
" ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler #Add intercept to prediction\n",
|
||||
" print(\"Values for own Ridge prediction\")\n",
|
||||
" print(ypredictOwnRidge)\n",
|
||||
"\n",
|
||||
" \n",
|
||||
"\n",
|
||||
" RegRidge = linear_model.Ridge(lmb)\n",
|
||||
" RegRidge.fit(X_train,y_train)\n",
|
||||
" ypredictRidge = RegRidge.predict(X_test)\n",
|
||||
" print(\"Values for SL Ridge prediction\")\n",
|
||||
" print(ypredictRidge)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
" MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n",
|
||||
" MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n",
|
||||
"\n",
|
||||
" print(\"Beta values for own Ridge implementation\")\n",
|
||||
" print(OwnRidgeBeta) #Intercept is given by mean of target variable\n",
|
||||
" print(\"Beta values for Scikit-Learn Ridge implementation\")\n",
|
||||
" print(RegRidge.coef_)\n",
|
||||
" print('Intercept from own implementation:')\n",
|
||||
" print(intercept_)\n",
|
||||
" print('Intercept from Scikit-Learn Ridge implementation')\n",
|
||||
" print(RegRidge.intercept_)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Now plot the results\n",
|
||||
"\n",
|
||||
"plt.figure()\n",
|
||||
"plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')\n",
|
||||
"plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')\n",
|
||||
"\n",
|
||||
"plt.xlabel('log10(lambda)')\n",
|
||||
"plt.ylabel('MSE')\n",
|
||||
"plt.legend()\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
"metadata": {
|
||||
"collapsed": false,
|
||||
"editable": true
|
||||
},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"import numpy as np\n",
|
||||
"import matplotlib.pyplot as plt\n",
|
||||
"\n",
|
||||
"from sklearn.linear_model import LinearRegression\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"np.random.seed(2021)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"def fit_beta(X, y):\n",
|
||||
" return np.linalg.pinv(X.T @ X) @ X.T @ y\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"true_beta = [2, 0.5, 3.7]\n",
|
||||
"\n",
|
||||
"x = np.linspace(0, 1, 11)\n",
|
||||
"y = np.sum(\n",
|
||||
" np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0\n",
|
||||
") + 0.1 * np.random.normal(size=len(x))\n",
|
||||
"\n",
|
||||
"degree = 3\n",
|
||||
"X = np.zeros((len(x), degree))\n",
|
||||
"\n",
|
||||
"# Include the intercept in the design matrix\n",
|
||||
"for p in range(degree):\n",
|
||||
" X[:, p] = x ** p\n",
|
||||
"\n",
|
||||
"beta = fit_beta(X, y)\n",
|
||||
"\n",
|
||||
"# Intercept is included in the design matrix\n",
|
||||
"clf = LinearRegression(fit_intercept=False).fit(X, y)\n",
|
||||
"\n",
|
||||
"print(f\"True beta: {true_beta}\")\n",
|
||||
"print(f\"Fitted beta: {beta}\")\n",
|
||||
"print(f\"Sklearn fitted beta: {clf.coef_}\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"plt.figure()\n",
|
||||
"plt.scatter(x, y, label=\"Data\")\n",
|
||||
"plt.plot(x, X @ beta, label=\"Fit\")\n",
|
||||
"plt.plot(x, clf.predict(X), label=\"Sklearn (fit_intercept=False)\")\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Do not include the intercept in the design matrix\n",
|
||||
"X = np.zeros((len(x), degree - 1))\n",
|
||||
"\n",
|
||||
"for p in range(degree - 1):\n",
|
||||
" X[:, p] = x ** (p + 1)\n",
|
||||
"\n",
|
||||
"# Intercept is not included in the design matrix\n",
|
||||
"clf = LinearRegression(fit_intercept=True).fit(X, y)\n",
|
||||
"\n",
|
||||
"# Use centered values for X and y when computing coefficients\n",
|
||||
"y_offset = np.average(y, axis=0)\n",
|
||||
"X_offset = np.average(X, axis=0)\n",
|
||||
"\n",
|
||||
"beta = fit_beta(X - X_offset, y - y_offset)\n",
|
||||
"intercept = np.mean(y_offset - X_offset @ beta)\n",
|
||||
"\n",
|
||||
"print(f\"Manual intercept: {intercept}\")\n",
|
||||
"print(f\"Fitted beta (sans intercept): {beta}\")\n",
|
||||
"print(f\"Sklearn intercept: {clf.intercept_}\")\n",
|
||||
"print(f\"Sklearn fitted beta (sans intercept): {clf.coef_}\")\n",
|
||||
"\n",
|
||||
"plt.plot(x, X @ beta + intercept, \"--\", label=\"Fit (manual intercept)\")\n",
|
||||
"plt.plot(x, clf.predict(X), \"--\", label=\"Sklearn (fit_intercept=True)\")\n",
|
||||
"plt.grid()\n",
|
||||
"plt.legend()\n",
|
||||
"\n",
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
@@ -1116,7 +1577,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"2\n",
|
||||
"3\n",
|
||||
"2\n",
|
||||
" \n",
|
||||
"<\n",
|
||||
|
||||
@@ -48,7 +48,7 @@ y_train_scaled = y_train - y_scaler #Remove the intercept from the training da
|
||||
p = Maxpolydegree-1
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 1
|
||||
nlambdas = 4
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
@@ -60,8 +60,6 @@ for i in range(nlambdas):
|
||||
print("Values for own Ridge prediction")
|
||||
print(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
|
||||
@@ -0,0 +1,100 @@
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn import linear_model
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
def R2(y_data, y_model):
|
||||
return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(315)
|
||||
|
||||
n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 5
|
||||
X = np.zeros((n,Maxpolydegree-1))
|
||||
|
||||
for degree in range(1,Maxpolydegree): #No intercept column
|
||||
X[:,degree-1] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
X_train_scaled = X_train - X_train_mean #Center by removing mean from each feature
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) #The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)
|
||||
y_train_scaled = y_train - y_scaler #Remove the intercept from the training data.
|
||||
|
||||
|
||||
p = Maxpolydegree-1
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 4
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-4, 1, nlambdas)
|
||||
for i in range(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ #Add intercept to prediction
|
||||
#EQUIVALENT PREDICTION:
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler #Add intercept to prediction
|
||||
print("Values for own Ridge prediction")
|
||||
print(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
print("Values for SL Ridge prediction")
|
||||
print(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
print("Beta values for own Ridge implementation")
|
||||
print(OwnRidgeBeta) #Intercept is given by mean of target variable
|
||||
print("Beta values for Scikit-Learn Ridge implementation")
|
||||
print(RegRidge.coef_)
|
||||
print('Intercept from own implementation:')
|
||||
print(intercept_)
|
||||
print('Intercept from Scikit-Learn Ridge implementation')
|
||||
print(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
# Now plot the results
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
|
||||
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
|
||||
+345
-135
@@ -4,138 +4,6 @@ DATE: today
|
||||
|
||||
|
||||
|
||||
I dont think its expected that you do this if it hasn't been gone through in the lectures or the curriculum, but anyways:
|
||||
|
||||
|
||||
|
||||
|
||||
Yes, it could be a bad idea to include the intercept column for the exact reason you stated. If no transformation is applied to your data, the intercept can be interpreted as the expected value of your target variable when all your predictors are put to zero. Therefore, whenever you cannot assume that the expected target variable is zero when all your predictors are zero, it could be a bad idea to apply a model which penalizes the intercept. Also, the analytical solution to the ridge regression coefficients (when not shrinking $$\beta_0$$) is derived under the assumption that both y and X are zero centered (mean subtracted). What you are doing is correct, but you should also zero center X (subtracting the mean of each column from the corresponding column).
|
||||
|
||||
If your predictors are of different scales, I would advice you to standardize X by subtracting the mean of each column from the corresponding column and dividing the column with its standard deviation. If you dont do this, you will give an "unfair" penalization of the parameters since their magnitude depends on the scale of their corresponding predictor. Suppose that you have an input variable "height". Human height might be measured in inches or meters or kilometers. If measured in kilometers, a standard linear regression model with this predictor would probably give a much bigger coefficient term, than if measured in millimeters. You may see how this could become a problem when considering the loss function for ridge regression.
|
||||
|
||||
|
||||
|
||||
|
||||
Remember that when you do any transformation to your dataset before training, the exact same transformation has to be applied to new data before making a prediction. In your case, this means:
|
||||
|
||||
#Model training:
|
||||
y_train_mean = np.mean(y_train)
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
X_train = X_train - X_train_mean
|
||||
y_train = y_train - y_train_mean
|
||||
|
||||
trained_model = some_model.fit(X_train,y_train)
|
||||
|
||||
#Model prediction:
|
||||
X_test = X_test - X_train_mean #Use mean from training data
|
||||
|
||||
|
||||
y_pred = trained_model(X_test)
|
||||
y_pred = y_pred + y_train_mean
|
||||
|
||||
|
||||
|
||||
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
|
||||
|
||||
|
||||
|
||||
The loss for ridge regression is:
|
||||
|
||||
|
||||
|
||||
|
||||
$$L(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2$$
|
||||
|
||||
|
||||
|
||||
|
||||
Notice that the intercept is left out of the L2 regularization term. $$X$$ does in this case not contain any intercept column. We want
|
||||
|
||||
$$ \frac{\partial L}{\partial \beta_j} = 0 $$
|
||||
|
||||
for all j, so lets start with $$\beta_0$$:
|
||||
|
||||
|
||||
|
||||
|
||||
$$\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) $$
|
||||
|
||||
|
||||
|
||||
|
||||
We want to solve
|
||||
|
||||
$$ -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0 $$
|
||||
|
||||
|
||||
|
||||
|
||||
which gives
|
||||
|
||||
$$ \sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p $$
|
||||
|
||||
|
||||
|
||||
|
||||
or
|
||||
|
||||
|
||||
|
||||
|
||||
$$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$$
|
||||
|
||||
|
||||
|
||||
|
||||
If we assume that every column of $$X$$ is centered, which we can do by subtracting the mean,
|
||||
|
||||
X = X - np.mean(X,axis=0)
|
||||
|
||||
the sum
|
||||
|
||||
$$ \sum_{i=1}^{n} X_{ip} $$
|
||||
|
||||
can be rewritten as
|
||||
|
||||
$$ \sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip}$$
|
||||
|
||||
$$ = \sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0 $$
|
||||
|
||||
|
||||
|
||||
|
||||
Finally we have
|
||||
|
||||
|
||||
|
||||
|
||||
$$n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$$
|
||||
|
||||
$$ \beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average} $$
|
||||
|
||||
|
||||
|
||||
|
||||
Replacing $$y_i$$ with $$y_i - \beta_0 = y_i - y_{average}$$ in the loss function will give us (written in vector notation)
|
||||
|
||||
$$L(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}$$
|
||||
|
||||
|
||||
|
||||
|
||||
which has the solution you stated
|
||||
|
||||
$$\beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}$$
|
||||
|
||||
|
||||
|
||||
|
||||
where $$\boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average}$$
|
||||
|
||||
and $$\tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} $$
|
||||
|
||||
|
||||
|
||||
!split
|
||||
@@ -369,10 +237,171 @@ plt.show()
|
||||
|
||||
|
||||
!split
|
||||
===== To think about =====
|
||||
===== To think about, first part =====
|
||||
|
||||
When you are comparing your own code with for example _Scikit-Learn_'s
|
||||
library, there are some things to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
|
||||
The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which may confuse us.
|
||||
|
||||
|
||||
|
||||
Yes, it could be a bad idea to include the intercept column for the
|
||||
exact reason you stated. If no transformation is applied to your data,
|
||||
the intercept can be interpreted as the expected value of your target
|
||||
variable when all your predictors are put to zero. Therefore, whenever
|
||||
you cannot assume that the expected target variable is zero when all
|
||||
your predictors are zero, it could be a bad idea to apply a model
|
||||
which penalizes the intercept. Also, the analytical solution to the
|
||||
ridge regression coefficients (when not shrinking $$\beta_0$$) is
|
||||
derived under the assumption that both y and X are zero centered (mean
|
||||
subtracted). What you are doing is correct, but you should also zero
|
||||
center X (subtracting the mean of each column from the corresponding
|
||||
column).
|
||||
|
||||
If your predictors are of different scales, I would advice you to
|
||||
standardize X by subtracting the mean of each column from the
|
||||
corresponding column and dividing the column with its standard
|
||||
deviation. If you dont do this, you will give an "unfair" penalization
|
||||
of the parameters since their magnitude depends on the scale of their
|
||||
corresponding predictor. Suppose that you have an input variable
|
||||
"height". Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters. You may see how
|
||||
this could become a problem when considering the loss function for
|
||||
ridge regression.
|
||||
|
||||
|
||||
|
||||
|
||||
Remember that when you do any transformation to your dataset before
|
||||
training, the exact same transformation has to be applied to new data
|
||||
before making a prediction. In your case, this means:
|
||||
|
||||
!bc pycod
|
||||
#Model training:
|
||||
y_train_mean = np.mean(y_train)
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
X_train = X_train - X_train_mean
|
||||
y_train = y_train - y_train_mean
|
||||
|
||||
trained_model = some_model.fit(X_train,y_train)
|
||||
|
||||
#Model prediction:
|
||||
X_test = X_test - X_train_mean #Use mean from training data
|
||||
y_pred = trained_model(X_test)
|
||||
y_pred = y_pred + y_train_mean
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
|
||||
|
||||
|
||||
|
||||
The cost/loss function for Ridge regression is:
|
||||
|
||||
|
||||
!bt
|
||||
\[
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
\]
|
||||
!et
|
||||
|
||||
|
||||
|
||||
|
||||
Notice that the intercept is left out of the $L_2$ regularization term. The design matrix
|
||||
$X$ does in this case not contain any intercept column. We want
|
||||
|
||||
!bt
|
||||
\[
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
\]
|
||||
!et
|
||||
|
||||
for all $j$, so lets start with $\beta_0$. This means that we have
|
||||
|
||||
!bt
|
||||
\[
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
\]
|
||||
!et
|
||||
|
||||
We want to solve
|
||||
!bt
|
||||
\[
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
\]
|
||||
!et
|
||||
|
||||
|
||||
which gives
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
\]
|
||||
!et
|
||||
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
|
||||
|
||||
|
||||
If we assume that every column of $X$ is centered, whic we can do by subtracting the mean,
|
||||
!bc pycod
|
||||
X = X - np.mean(X,axis=0)
|
||||
!ec
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
can be rewritten as
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
\]
|
||||
!et
|
||||
resulting in
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
\]
|
||||
!et
|
||||
|
||||
|
||||
Finally we have
|
||||
!bt
|
||||
\[
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
\]
|
||||
!et
|
||||
or
|
||||
!bt
|
||||
\[
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
\]
|
||||
!et
|
||||
|
||||
Replacing $y_i$ with $y_i - \beta_0 = y_i - y_{average}$ in the loss function will give us (in vector-matrix disguise)
|
||||
!bt
|
||||
\[
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
\]
|
||||
!et
|
||||
|
||||
which has the solution
|
||||
|
||||
$\beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}$.
|
||||
where $\boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average}$
|
||||
and $\tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj}$.
|
||||
|
||||
|
||||
|
||||
When you are comparing your own code with for example _Scikit-Learn_'s library, there are some minor things to keep in mind.
|
||||
The example here shows how one can keep the intercept in order to compare own code.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
@@ -463,6 +492,187 @@ plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn import linear_model
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
def R2(y_data, y_model):
|
||||
return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(315)
|
||||
|
||||
n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 5
|
||||
X = np.zeros((n,Maxpolydegree-1))
|
||||
|
||||
for degree in range(1,Maxpolydegree): #No intercept column
|
||||
X[:,degree-1] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
X_train_scaled = X_train - X_train_mean #Center by removing mean from each feature
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) #The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)
|
||||
y_train_scaled = y_train - y_scaler #Remove the intercept from the training data.
|
||||
|
||||
|
||||
p = Maxpolydegree-1
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 4
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-4, 1, nlambdas)
|
||||
for i in range(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ #Add intercept to prediction
|
||||
#EQUIVALENT PREDICTION:
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler #Add intercept to prediction
|
||||
print("Values for own Ridge prediction")
|
||||
print(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
print("Values for SL Ridge prediction")
|
||||
print(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
print("Beta values for own Ridge implementation")
|
||||
print(OwnRidgeBeta) #Intercept is given by mean of target variable
|
||||
print("Beta values for Scikit-Learn Ridge implementation")
|
||||
print(RegRidge.coef_)
|
||||
print('Intercept from own implementation:')
|
||||
print(intercept_)
|
||||
print('Intercept from Scikit-Learn Ridge implementation')
|
||||
print(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
# Now plot the results
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
|
||||
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
|
||||
|
||||
!ec
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
from sklearn.linear_model import LinearRegression
|
||||
|
||||
|
||||
np.random.seed(2021)
|
||||
|
||||
|
||||
def fit_beta(X, y):
|
||||
return np.linalg.pinv(X.T @ X) @ X.T @ y
|
||||
|
||||
|
||||
true_beta = [2, 0.5, 3.7]
|
||||
|
||||
x = np.linspace(0, 1, 11)
|
||||
y = np.sum(
|
||||
np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0
|
||||
) + 0.1 * np.random.normal(size=len(x))
|
||||
|
||||
degree = 3
|
||||
X = np.zeros((len(x), degree))
|
||||
|
||||
# Include the intercept in the design matrix
|
||||
for p in range(degree):
|
||||
X[:, p] = x ** p
|
||||
|
||||
beta = fit_beta(X, y)
|
||||
|
||||
# Intercept is included in the design matrix
|
||||
clf = LinearRegression(fit_intercept=False).fit(X, y)
|
||||
|
||||
print(f"True beta: {true_beta}")
|
||||
print(f"Fitted beta: {beta}")
|
||||
print(f"Sklearn fitted beta: {clf.coef_}")
|
||||
|
||||
|
||||
plt.figure()
|
||||
plt.scatter(x, y, label="Data")
|
||||
plt.plot(x, X @ beta, label="Fit")
|
||||
plt.plot(x, clf.predict(X), label="Sklearn (fit_intercept=False)")
|
||||
|
||||
|
||||
# Do not include the intercept in the design matrix
|
||||
X = np.zeros((len(x), degree - 1))
|
||||
|
||||
for p in range(degree - 1):
|
||||
X[:, p] = x ** (p + 1)
|
||||
|
||||
# Intercept is not included in the design matrix
|
||||
clf = LinearRegression(fit_intercept=True).fit(X, y)
|
||||
|
||||
# Use centered values for X and y when computing coefficients
|
||||
y_offset = np.average(y, axis=0)
|
||||
X_offset = np.average(X, axis=0)
|
||||
|
||||
beta = fit_beta(X - X_offset, y - y_offset)
|
||||
intercept = np.mean(y_offset - X_offset @ beta)
|
||||
|
||||
print(f"Manual intercept: {intercept}")
|
||||
print(f"Fitted beta (sans intercept): {beta}")
|
||||
print(f"Sklearn intercept: {clf.intercept_}")
|
||||
print(f"Sklearn fitted beta (sans intercept): {clf.coef_}")
|
||||
|
||||
plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
|
||||
plt.plot(x, clf.predict(X), "--", label="Sklearn (fit_intercept=True)")
|
||||
plt.grid()
|
||||
plt.legend()
|
||||
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
!split
|
||||
===== More complicated Example: The Ising model =====
|
||||
|
||||
|
||||
Reference in New Issue
Block a user