914 lines
62 KiB
HTML
914 lines
62 KiB
HTML
<!--
|
|
HTML file automatically generated from DocOnce source
|
|
(https://github.com/doconce/doconce/)
|
|
doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=default --html_admon=bootstrap_panel --html_output=week35-bs --no_mako
|
|
-->
|
|
<html>
|
|
<head>
|
|
<meta http-equiv="Content-Type" content="text/html; charset=utf-8" />
|
|
<meta name="generator" content="DocOnce: https://github.com/doconce/doconce/" />
|
|
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
<meta name="description" content="Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression">
|
|
<title>Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression</title>
|
|
<!-- Bootstrap style: bootstrap -->
|
|
<!-- doconce format html week35.do.txt --html_style=bootstrap --pygments_html_style=default --html_admon=bootstrap_panel --html_output=week35-bs --no_mako -->
|
|
<link href="https://netdna.bootstrapcdn.com/bootstrap/3.1.1/css/bootstrap.min.css" rel="stylesheet">
|
|
<!-- not necessary
|
|
<link href="https://netdna.bootstrapcdn.com/font-awesome/4.0.3/css/font-awesome.css" rel="stylesheet">
|
|
-->
|
|
<style type="text/css">
|
|
/* Add scrollbar to dropdown menus in bootstrap navigation bar */
|
|
.dropdown-menu {
|
|
height: auto;
|
|
max-height: 400px;
|
|
overflow-x: hidden;
|
|
}
|
|
/* Adds an invisible element before each target to offset for the navigation
|
|
bar */
|
|
.anchor::before {
|
|
content:"";
|
|
display:block;
|
|
height:50px; /* fixed header height for style bootstrap */
|
|
margin:-50px 0 0; /* negative fixed header height */
|
|
}
|
|
</style>
|
|
</head>
|
|
|
|
<!-- tocinfo
|
|
{'highest level': 2,
|
|
'sections': [('Plans for week 35', 2, None, 'plans-for-week-35'),
|
|
('Reading recommendations:', 3, None, 'reading-recommendations'),
|
|
('Reminder from last week', 2, None, 'reminder-from-last-week'),
|
|
('The equations for ordinary least squares',
|
|
2,
|
|
None,
|
|
'the-equations-for-ordinary-least-squares'),
|
|
('The cost/loss function', 2, None, 'the-cost-loss-function'),
|
|
('Interpretations and optimizing our parameters',
|
|
2,
|
|
None,
|
|
'interpretations-and-optimizing-our-parameters'),
|
|
('Interpretations and optimizing our parameters',
|
|
2,
|
|
None,
|
|
'interpretations-and-optimizing-our-parameters'),
|
|
('Some useful matrix and vector expressions',
|
|
2,
|
|
None,
|
|
'some-useful-matrix-and-vector-expressions'),
|
|
('The Jacobian', 2, None, 'the-jacobian'),
|
|
('Derivatives, example 1', 2, None, 'derivatives-example-1'),
|
|
('Example 2', 2, None, 'example-2'),
|
|
('Example 3', 2, None, 'example-3'),
|
|
('Example 4', 2, None, 'example-4'),
|
|
('The mean squared error and its derivative',
|
|
2,
|
|
None,
|
|
'the-mean-squared-error-and-its-derivative'),
|
|
('Meet the Hessian Matrix', 2, None, 'meet-the-hessian-matrix'),
|
|
('Interpretations and optimizing our parameters',
|
|
2,
|
|
None,
|
|
'interpretations-and-optimizing-our-parameters'),
|
|
('Example relevant for the exercises',
|
|
2,
|
|
None,
|
|
'example-relevant-for-the-exercises'),
|
|
('Own code for Ordinary Least Squares',
|
|
2,
|
|
None,
|
|
'own-code-for-ordinary-least-squares'),
|
|
('Adding error analysis and training set up',
|
|
2,
|
|
None,
|
|
'adding-error-analysis-and-training-set-up'),
|
|
('Splitting our Data in Training and Test data',
|
|
2,
|
|
None,
|
|
'splitting-our-data-in-training-and-test-data'),
|
|
('The complete code with a simple data set',
|
|
2,
|
|
None,
|
|
'the-complete-code-with-a-simple-data-set'),
|
|
('Making your own test-train splitting',
|
|
2,
|
|
None,
|
|
'making-your-own-test-train-splitting'),
|
|
('Reducing the number of degrees of freedom, overarching view',
|
|
2,
|
|
None,
|
|
'reducing-the-number-of-degrees-of-freedom-overarching-view'),
|
|
('Preprocessing our data', 2, None, 'preprocessing-our-data'),
|
|
('Functionality in Scikit-Learn',
|
|
2,
|
|
None,
|
|
'functionality-in-scikit-learn'),
|
|
('More preprocessing', 2, None, 'more-preprocessing'),
|
|
('Frequently used scaling functions',
|
|
2,
|
|
None,
|
|
'frequently-used-scaling-functions'),
|
|
('Example of own Standard scaling',
|
|
2,
|
|
None,
|
|
'example-of-own-standard-scaling'),
|
|
('Min-Max Scaling', 2, None, 'min-max-scaling'),
|
|
('Testing the Means Squared Error as function of Complexity',
|
|
2,
|
|
None,
|
|
'testing-the-means-squared-error-as-function-of-complexity'),
|
|
('Mathematical Interpretation of Ordinary Least Squares',
|
|
2,
|
|
None,
|
|
'mathematical-interpretation-of-ordinary-least-squares'),
|
|
('Residual Error', 2, None, 'residual-error'),
|
|
('Simple case', 2, None, 'simple-case'),
|
|
('The singular value decomposition',
|
|
2,
|
|
None,
|
|
'the-singular-value-decomposition'),
|
|
('Linear Regression Problems',
|
|
2,
|
|
None,
|
|
'linear-regression-problems'),
|
|
('Fixing the singularity', 2, None, 'fixing-the-singularity'),
|
|
('Ridge and LASSO Regression',
|
|
2,
|
|
None,
|
|
'ridge-and-lasso-regression'),
|
|
('Deriving the Ridge Regression Equations',
|
|
2,
|
|
None,
|
|
'deriving-the-ridge-regression-equations'),
|
|
('Basic math of the SVD', 2, None, 'basic-math-of-the-svd'),
|
|
('The SVD, a Fantastic Algorithm',
|
|
2,
|
|
None,
|
|
'the-svd-a-fantastic-algorithm'),
|
|
('Economy-size SVD', 2, None, 'economy-size-svd'),
|
|
('Codes for the SVD', 2, None, 'codes-for-the-svd'),
|
|
('Note about SVD Calculations',
|
|
2,
|
|
None,
|
|
'note-about-svd-calculations'),
|
|
('Mathematics of the SVD and implications',
|
|
2,
|
|
None,
|
|
'mathematics-of-the-svd-and-implications'),
|
|
('Example Matrix', 2, None, 'example-matrix'),
|
|
('Setting up the Matrix to be inverted',
|
|
2,
|
|
None,
|
|
'setting-up-the-matrix-to-be-inverted'),
|
|
('Further properties (important for our analyses later)',
|
|
2,
|
|
None,
|
|
'further-properties-important-for-our-analyses-later'),
|
|
('Meet the Covariance Matrix',
|
|
2,
|
|
None,
|
|
'meet-the-covariance-matrix'),
|
|
('Introducing the Covariance and Correlation functions',
|
|
2,
|
|
None,
|
|
'introducing-the-covariance-and-correlation-functions'),
|
|
('Covariance and Correlation Matrix',
|
|
2,
|
|
None,
|
|
'covariance-and-correlation-matrix'),
|
|
('Correlation Function and Design/Feature Matrix',
|
|
2,
|
|
None,
|
|
'correlation-function-and-design-feature-matrix'),
|
|
('Covariance Matrix Examples',
|
|
2,
|
|
None,
|
|
'covariance-matrix-examples'),
|
|
('Correlation Matrix', 2, None, 'correlation-matrix'),
|
|
('Correlation Matrix with Pandas',
|
|
2,
|
|
None,
|
|
'correlation-matrix-with-pandas'),
|
|
('Rewriting the Covariance and/or Correlation Matrix',
|
|
2,
|
|
None,
|
|
'rewriting-the-covariance-and-or-correlation-matrix'),
|
|
('Linking with the SVD', 2, None, 'linking-with-the-svd'),
|
|
('What does it mean?', 2, None, 'what-does-it-mean'),
|
|
('And finally $\\boldsymbol{X}\\boldsymbol{X}^T$',
|
|
2,
|
|
None,
|
|
'and-finally-boldsymbol-x-boldsymbol-x-t'),
|
|
('Back to Ridge and LASSO Regression',
|
|
2,
|
|
None,
|
|
'back-to-ridge-and-lasso-regression'),
|
|
('Interpreting the Ridge results',
|
|
2,
|
|
None,
|
|
'interpreting-the-ridge-results'),
|
|
('More interpretations', 2, None, 'more-interpretations'),
|
|
('Deriving the Lasso Regression Equations',
|
|
2,
|
|
None,
|
|
'deriving-the-lasso-regression-equations'),
|
|
('Material for exercises week 35',
|
|
2,
|
|
None,
|
|
'material-for-exercises-week-35'),
|
|
('Important technicalities: More on Rescaling data',
|
|
2,
|
|
None,
|
|
'important-technicalities-more-on-rescaling-data')]}
|
|
end of tocinfo -->
|
|
|
|
<body>
|
|
|
|
|
|
|
|
<script type="text/x-mathjax-config">
|
|
MathJax.Hub.Config({
|
|
TeX: {
|
|
equationNumbers: { autoNumber: "none" },
|
|
extensions: ["AMSmath.js", "AMSsymbols.js", "autobold.js", "color.js"]
|
|
}
|
|
});
|
|
</script>
|
|
<script type="text/javascript" async
|
|
src="https://cdnjs.cloudflare.com/ajax/libs/mathjax/2.7.1/MathJax.js?config=TeX-AMS-MML_HTMLorMML">
|
|
</script>
|
|
|
|
|
|
<!-- Bootstrap navigation bar -->
|
|
<div class="navbar navbar-default navbar-fixed-top">
|
|
<div class="navbar-header">
|
|
<button type="button" class="navbar-toggle" data-toggle="collapse" data-target=".navbar-responsive-collapse">
|
|
<span class="icon-bar"></span>
|
|
<span class="icon-bar"></span>
|
|
<span class="icon-bar"></span>
|
|
</button>
|
|
<a class="navbar-brand" href="week35-bs.html">Week 35: From Ordinary Linear Regression to Ridge and Lasso Regression</a>
|
|
</div>
|
|
<div class="navbar-collapse collapse navbar-responsive-collapse">
|
|
<ul class="nav navbar-nav navbar-right">
|
|
<li class="dropdown">
|
|
<a href="#" class="dropdown-toggle" data-toggle="dropdown">Contents <b class="caret"></b></a>
|
|
<ul class="dropdown-menu">
|
|
<!-- navigation toc: --> <li><a href="._week35-bs001.html#plans-for-week-35" style="font-size: 80%;"><b>Plans for week 35</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs001.html#reading-recommendations" style="font-size: 80%;"> Reading recommendations:</a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs002.html#reminder-from-last-week" style="font-size: 80%;"><b>Reminder from last week</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs003.html#the-equations-for-ordinary-least-squares" style="font-size: 80%;"><b>The equations for ordinary least squares</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs004.html#the-cost-loss-function" style="font-size: 80%;"><b>The cost/loss function</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs015.html#interpretations-and-optimizing-our-parameters" style="font-size: 80%;"><b>Interpretations and optimizing our parameters</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs015.html#interpretations-and-optimizing-our-parameters" style="font-size: 80%;"><b>Interpretations and optimizing our parameters</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs007.html#some-useful-matrix-and-vector-expressions" style="font-size: 80%;"><b>Some useful matrix and vector expressions</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs008.html#the-jacobian" style="font-size: 80%;"><b>The Jacobian</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs009.html#derivatives-example-1" style="font-size: 80%;"><b>Derivatives, example 1</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs010.html#example-2" style="font-size: 80%;"><b>Example 2</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs011.html#example-3" style="font-size: 80%;"><b>Example 3</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs012.html#example-4" style="font-size: 80%;"><b>Example 4</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs013.html#the-mean-squared-error-and-its-derivative" style="font-size: 80%;"><b>The mean squared error and its derivative</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs014.html#meet-the-hessian-matrix" style="font-size: 80%;"><b>Meet the Hessian Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs015.html#interpretations-and-optimizing-our-parameters" style="font-size: 80%;"><b>Interpretations and optimizing our parameters</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs016.html#example-relevant-for-the-exercises" style="font-size: 80%;"><b>Example relevant for the exercises</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs017.html#own-code-for-ordinary-least-squares" style="font-size: 80%;"><b>Own code for Ordinary Least Squares</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs018.html#adding-error-analysis-and-training-set-up" style="font-size: 80%;"><b>Adding error analysis and training set up</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs019.html#splitting-our-data-in-training-and-test-data" style="font-size: 80%;"><b>Splitting our Data in Training and Test data</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs020.html#the-complete-code-with-a-simple-data-set" style="font-size: 80%;"><b>The complete code with a simple data set</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs021.html#making-your-own-test-train-splitting" style="font-size: 80%;"><b>Making your own test-train splitting</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs022.html#reducing-the-number-of-degrees-of-freedom-overarching-view" style="font-size: 80%;"><b>Reducing the number of degrees of freedom, overarching view</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs023.html#preprocessing-our-data" style="font-size: 80%;"><b>Preprocessing our data</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs024.html#functionality-in-scikit-learn" style="font-size: 80%;"><b>Functionality in Scikit-Learn</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs025.html#more-preprocessing" style="font-size: 80%;"><b>More preprocessing</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs026.html#frequently-used-scaling-functions" style="font-size: 80%;"><b>Frequently used scaling functions</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs027.html#example-of-own-standard-scaling" style="font-size: 80%;"><b>Example of own Standard scaling</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs028.html#min-max-scaling" style="font-size: 80%;"><b>Min-Max Scaling</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs029.html#testing-the-means-squared-error-as-function-of-complexity" style="font-size: 80%;"><b>Testing the Means Squared Error as function of Complexity</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs030.html#mathematical-interpretation-of-ordinary-least-squares" style="font-size: 80%;"><b>Mathematical Interpretation of Ordinary Least Squares</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs031.html#residual-error" style="font-size: 80%;"><b>Residual Error</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs032.html#simple-case" style="font-size: 80%;"><b>Simple case</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs033.html#the-singular-value-decomposition" style="font-size: 80%;"><b>The singular value decomposition</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs034.html#linear-regression-problems" style="font-size: 80%;"><b>Linear Regression Problems</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs035.html#fixing-the-singularity" style="font-size: 80%;"><b>Fixing the singularity</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs036.html#ridge-and-lasso-regression" style="font-size: 80%;"><b>Ridge and LASSO Regression</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs037.html#deriving-the-ridge-regression-equations" style="font-size: 80%;"><b>Deriving the Ridge Regression Equations</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs038.html#basic-math-of-the-svd" style="font-size: 80%;"><b>Basic math of the SVD</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs039.html#the-svd-a-fantastic-algorithm" style="font-size: 80%;"><b>The SVD, a Fantastic Algorithm</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs040.html#economy-size-svd" style="font-size: 80%;"><b>Economy-size SVD</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs041.html#codes-for-the-svd" style="font-size: 80%;"><b>Codes for the SVD</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs042.html#note-about-svd-calculations" style="font-size: 80%;"><b>Note about SVD Calculations</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs043.html#mathematics-of-the-svd-and-implications" style="font-size: 80%;"><b>Mathematics of the SVD and implications</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs044.html#example-matrix" style="font-size: 80%;"><b>Example Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs045.html#setting-up-the-matrix-to-be-inverted" style="font-size: 80%;"><b>Setting up the Matrix to be inverted</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs046.html#further-properties-important-for-our-analyses-later" style="font-size: 80%;"><b>Further properties (important for our analyses later)</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs047.html#meet-the-covariance-matrix" style="font-size: 80%;"><b>Meet the Covariance Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs048.html#introducing-the-covariance-and-correlation-functions" style="font-size: 80%;"><b>Introducing the Covariance and Correlation functions</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs049.html#covariance-and-correlation-matrix" style="font-size: 80%;"><b>Covariance and Correlation Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs050.html#correlation-function-and-design-feature-matrix" style="font-size: 80%;"><b>Correlation Function and Design/Feature Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs051.html#covariance-matrix-examples" style="font-size: 80%;"><b>Covariance Matrix Examples</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs052.html#correlation-matrix" style="font-size: 80%;"><b>Correlation Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs053.html#correlation-matrix-with-pandas" style="font-size: 80%;"><b>Correlation Matrix with Pandas</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs054.html#rewriting-the-covariance-and-or-correlation-matrix" style="font-size: 80%;"><b>Rewriting the Covariance and/or Correlation Matrix</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs055.html#linking-with-the-svd" style="font-size: 80%;"><b>Linking with the SVD</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs056.html#what-does-it-mean" style="font-size: 80%;"><b>What does it mean?</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs057.html#and-finally-boldsymbol-x-boldsymbol-x-t" style="font-size: 80%;"><b>And finally \( \boldsymbol{X}\boldsymbol{X}^T \)</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs058.html#back-to-ridge-and-lasso-regression" style="font-size: 80%;"><b>Back to Ridge and LASSO Regression</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs059.html#interpreting-the-ridge-results" style="font-size: 80%;"><b>Interpreting the Ridge results</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs060.html#more-interpretations" style="font-size: 80%;"><b>More interpretations</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="._week35-bs061.html#deriving-the-lasso-regression-equations" style="font-size: 80%;"><b>Deriving the Lasso Regression Equations</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="#material-for-exercises-week-35" style="font-size: 80%;"><b>Material for exercises week 35</b></a></li>
|
|
<!-- navigation toc: --> <li><a href="#important-technicalities-more-on-rescaling-data" style="font-size: 80%;"><b>Important technicalities: More on Rescaling data</b></a></li>
|
|
|
|
</ul>
|
|
</li>
|
|
</ul>
|
|
</div>
|
|
</div>
|
|
</div> <!-- end of navigation bar -->
|
|
<div class="container">
|
|
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
|
<a name="part0062"></a>
|
|
<!-- !split -->
|
|
<h2 id="material-for-exercises-week-35" class="anchor">Material for exercises week 35 </h2>
|
|
<h2 id="important-technicalities-more-on-rescaling-data" class="anchor">Important technicalities: More on Rescaling data </h2>
|
|
|
|
<p>When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
|
library, there are some technicalities to keep in mind. The examples
|
|
here demonstrate some of these aspects with potential pitfalls.
|
|
</p>
|
|
|
|
<p>The discussion here focuses on the role of the intercept, how we can
|
|
set up the design matrix, what scaling we should use and other topics
|
|
which tend confuse us.
|
|
</p>
|
|
|
|
<p>The intercept can be interpreted as the expected value of our
|
|
target/output variables when all other predictors are set to zero.
|
|
Thus, if we cannot assume that the expected outputs/targets are zero
|
|
when all predictors are zero (the columns in the design matrix), it
|
|
may be a bad idea to implement a model which penalizes the intercept.
|
|
Furthermore, in for example Ridge and Lasso regression, the default solutions
|
|
from the library <b>Scikit-Learn</b> (when not shrinking \( \beta_0 \)) for the unknown parameters
|
|
\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and
|
|
\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values.
|
|
</p>
|
|
|
|
<p>If our predictors represent different scales, then it is important to
|
|
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
|
column from the corresponding column and dividing the column with its
|
|
standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,
|
|
the results may differ.
|
|
</p>
|
|
|
|
<p>The
|
|
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_self">Standardscaler</a>
|
|
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
|
have been studying in our various examples, the data are in many cases
|
|
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
|
survey of your data, with a critical assessment of them in case you need to scale the data.
|
|
</p>
|
|
|
|
<p>If you need to scale the data, not doing so will give an <em>unfair</em>
|
|
penalization of the parameters since their magnitude depends on the
|
|
scale of their corresponding predictor.
|
|
</p>
|
|
|
|
<p>The <b>Scikit-Learn</b> site <a href="https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section" target="_self"><tt>https://scikit-learn.org/stable/auto_examples/preprocessing/plot_all_scaling.html#plot-all-scaling-standard-scaler-section</tt></a> has a good discussion of different ways of preprocessing data.</p>
|
|
|
|
<p>Suppose as an example that you
|
|
you have an input variable given by the heights of different persons.
|
|
Human height might be measured in inches or meters or
|
|
kilometers. If measured in kilometers, a standard linear regression
|
|
model with this predictor would probably give a much bigger
|
|
coefficient term, than if measured in millimeters.
|
|
This can clearly lead to problems in evaluating the cost/loss functions.
|
|
</p>
|
|
|
|
<p>Keep in mind that when you transform your data set before training a model, the same transformation needs to be done
|
|
on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #BA2121; font-style: italic">"""</span>
|
|
<span style="color: #BA2121; font-style: italic">#Model training, we compute the mean value of y and X</span>
|
|
<span style="color: #BA2121; font-style: italic">y_train_mean = np.mean(y_train)</span>
|
|
<span style="color: #BA2121; font-style: italic">X_train_mean = np.mean(X_train,axis=0)</span>
|
|
<span style="color: #BA2121; font-style: italic">X_train = X_train - X_train_mean</span>
|
|
<span style="color: #BA2121; font-style: italic">y_train = y_train - y_train_mean</span>
|
|
|
|
<span style="color: #BA2121; font-style: italic"># The we fit our model with the training data</span>
|
|
<span style="color: #BA2121; font-style: italic">trained_model = some_model.fit(X_train,y_train)</span>
|
|
|
|
|
|
<span style="color: #BA2121; font-style: italic">#Model prediction, we need also to transform our data set used for the prediction.</span>
|
|
<span style="color: #BA2121; font-style: italic">X_test = X_test - X_train_mean #Use mean from training data</span>
|
|
<span style="color: #BA2121; font-style: italic">y_pred = trained_model(X_test)</span>
|
|
<span style="color: #BA2121; font-style: italic">y_pred = y_pred + y_train_mean</span>
|
|
<span style="color: #BA2121; font-style: italic">"""</span>
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>Let us try to understand what this may imply mathematically when we
|
|
subtract the mean values, also known as <em>zero centering</em>. For
|
|
simplicity, we will focus on ordinary regression, as done in the above example.
|
|
</p>
|
|
|
|
<p>The cost/loss function for regression is</p>
|
|
$$
|
|
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,.
|
|
$$
|
|
|
|
<p>Recall also that we use the squared value. This expression can lead to an
|
|
increased penalty for higher differences between predicted and
|
|
output/target values.
|
|
</p>
|
|
|
|
<p>What we have done is to single out the \( \beta_0 \) term in the
|
|
definition of the mean squared error (MSE). The design matrix \( X \)
|
|
does in this case not contain any intercept column. When we take the
|
|
derivative with respect to \( \beta_0 \), we want the derivative to obey
|
|
</p>
|
|
|
|
$$
|
|
\frac{\partial C}{\partial \beta_j} = 0,
|
|
$$
|
|
|
|
<p>for all \( j \). For \( \beta_0 \) we have</p>
|
|
|
|
$$
|
|
\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right).
|
|
$$
|
|
|
|
<p>Multiplying away the constant \( 2/n \), we obtain</p>
|
|
$$
|
|
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
|
$$
|
|
|
|
<p>Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \).
|
|
Our result for \( \beta_0 \) simplifies then to
|
|
</p>
|
|
$$
|
|
n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1.
|
|
$$
|
|
|
|
<p>We obtain then</p>
|
|
$$
|
|
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}.
|
|
$$
|
|
|
|
<p>If we define</p>
|
|
$$
|
|
\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1},
|
|
$$
|
|
|
|
<p>and the mean value of the outputs as</p>
|
|
$$
|
|
\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i,
|
|
$$
|
|
|
|
<p>we have</p>
|
|
$$
|
|
\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}.
|
|
$$
|
|
|
|
<p>In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have</p>
|
|
$$
|
|
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j.
|
|
$$
|
|
|
|
<p>We can rewrite the latter equation as</p>
|
|
$$
|
|
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j,
|
|
$$
|
|
|
|
<p>where we have defined</p>
|
|
$$
|
|
\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij},
|
|
$$
|
|
|
|
<p>the mean value for all elements of the column vector \( \boldsymbol{x}_j \).</p>
|
|
|
|
<p>Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)</p>
|
|
$$
|
|
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}).
|
|
$$
|
|
|
|
<p>If we minimize with respect to \( \boldsymbol{\beta} \) we have then</p>
|
|
|
|
$$
|
|
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}},
|
|
$$
|
|
|
|
<p>where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
|
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \).
|
|
</p>
|
|
|
|
<p>For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then</p>
|
|
$$
|
|
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}.
|
|
$$
|
|
|
|
<p>What does this mean? And why do we insist on all this? Let us look at some examples.</p>
|
|
|
|
<p>This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
|
Note also that we do not split the data into training and test.
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression
|
|
|
|
|
|
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">2021</span>)
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
|
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
|
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">fit_beta</span>(X, y):
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X) <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
|
|
|
|
|
true_beta <span style="color: #666666">=</span> [<span style="color: #666666">2</span>, <span style="color: #666666">0.5</span>, <span style="color: #666666">3.7</span>]
|
|
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>, <span style="color: #666666">1</span>, <span style="color: #666666">11</span>)
|
|
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(
|
|
np<span style="color: #666666">.</span>asarray([x <span style="color: #666666">**</span> p <span style="color: #666666">*</span> b <span style="color: #008000; font-weight: bold">for</span> p, b <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(true_beta)]), axis<span style="color: #666666">=0</span>
|
|
) <span style="color: #666666">+</span> <span style="color: #666666">0.1</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(size<span style="color: #666666">=</span><span style="color: #008000">len</span>(x))
|
|
|
|
degree <span style="color: #666666">=</span> <span style="color: #666666">3</span>
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree))
|
|
|
|
<span style="color: #408080; font-style: italic"># Include the intercept in the design matrix</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree):
|
|
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> p
|
|
|
|
beta <span style="color: #666666">=</span> fit_beta(X, y)
|
|
|
|
<span style="color: #408080; font-style: italic"># Intercept is included in the design matrix</span>
|
|
skl <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)<span style="color: #666666">.</span>fit(X, y)
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"True beta: </span><span style="color: #BB6688; font-weight: bold">{</span>true_beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
ypredictOwn <span style="color: #666666">=</span> X <span style="color: #666666">@</span> beta
|
|
ypredictSKL <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>predict(X)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with intercept column"</span>)
|
|
<span style="color: #008000">print</span>(MSE(y,ypredictOwn))
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with intercept column from SKL"</span>)
|
|
<span style="color: #008000">print</span>(MSE(y,ypredictSKL))
|
|
|
|
|
|
plt<span style="color: #666666">.</span>figure()
|
|
plt<span style="color: #666666">.</span>scatter(x, y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Data"</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit"</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, skl<span style="color: #666666">.</span>predict(X), label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=False)"</span>)
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># Do not include the intercept in the design matrix</span>
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>))
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>):
|
|
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> (p <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
|
|
|
<span style="color: #408080; font-style: italic"># Intercept is not included in the design matrix</span>
|
|
skl <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)<span style="color: #666666">.</span>fit(X, y)
|
|
|
|
<span style="color: #408080; font-style: italic"># Use centered values for X and y when computing coefficients</span>
|
|
y_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(y, axis<span style="color: #666666">=0</span>)
|
|
X_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(X, axis<span style="color: #666666">=0</span>)
|
|
|
|
beta <span style="color: #666666">=</span> fit_beta(X <span style="color: #666666">-</span> X_offset, y <span style="color: #666666">-</span> y_offset)
|
|
intercept <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_offset <span style="color: #666666">-</span> X_offset <span style="color: #666666">@</span> beta)
|
|
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Manual intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>intercept<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta (without intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>intercept_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta (without intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
|
ypredictOwn <span style="color: #666666">=</span> X <span style="color: #666666">@</span> beta
|
|
ypredictSKL <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>predict(X)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with Manual intercept"</span>)
|
|
<span style="color: #008000">print</span>(MSE(y,ypredictOwn<span style="color: #666666">+</span>intercept))
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with Sklearn intercept"</span>)
|
|
<span style="color: #008000">print</span>(MSE(y,ypredictSKL))
|
|
|
|
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta <span style="color: #666666">+</span> intercept, <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit (manual intercept)"</span>)
|
|
plt<span style="color: #666666">.</span>plot(x, skl<span style="color: #666666">.</span>predict(X), <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=True)"</span>)
|
|
plt<span style="color: #666666">.</span>grid()
|
|
plt<span style="color: #666666">.</span>legend()
|
|
|
|
plt<span style="color: #666666">.</span>show()
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>The intercept is the value of our output/target variable
|
|
when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
|
|
</p>
|
|
|
|
<p>Printing the MSE, we see first that both methods give the same MSE, as
|
|
they should. However, when we move to for example Ridge regression,
|
|
the way we treat the intercept may give a larger or smaller MSE,
|
|
meaning that the MSE can be penalized by the value of the
|
|
intercept. Not including the intercept in the fit, means that the
|
|
regularization term does not include \( \beta_0 \). For different values
|
|
of \( \lambda \), this may lead to different MSE values.
|
|
</p>
|
|
|
|
<p>To remind the reader, the regularization term, with the intercept in Ridge regression, is given by</p>
|
|
$$
|
|
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2,
|
|
$$
|
|
|
|
<p>but when we take out the intercept, this equation becomes</p>
|
|
$$
|
|
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2.
|
|
$$
|
|
|
|
<p>For Lasso regression we have</p>
|
|
$$
|
|
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert.
|
|
$$
|
|
|
|
<p>It means that, when scaling the design matrix and the outputs/targets,
|
|
by subtracting the mean values, we have an optimization problem which
|
|
is not penalized by the intercept. The MSE value can then be smaller
|
|
since it focuses only on the remaining quantities. If we however bring
|
|
back the intercept, we will get a MSE which then contains the
|
|
intercept.
|
|
</p>
|
|
|
|
<p>Armed with this wisdom, we attempt first to simply set the intercept equal to <b>False</b> in our implementation of Ridge regression for our well-known vanilla data set.</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
|
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
|
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
|
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
|
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
|
|
|
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree))
|
|
<span style="color: #408080; font-style: italic">#We include explicitely the intercept column</span>
|
|
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Maxpolydegree):
|
|
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
|
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
|
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
|
|
|
p <span style="color: #666666">=</span> Maxpolydegree
|
|
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
|
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
|
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">6</span>
|
|
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
|
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
|
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">2</span>, nlambdas)
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
|
lmb <span style="color: #666666">=</span> lambdas[i]
|
|
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
|
<span style="color: #408080; font-style: italic"># Note: we include the intercept column and no scaling</span>
|
|
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb,fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)
|
|
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
|
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
|
ytildeOwnRidge <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OwnRidgeBeta
|
|
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta
|
|
ytildeRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_train)
|
|
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
|
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
|
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(OwnRidgeBeta)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for own Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(MSEOwnRidgePredict[i])
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(MSERidgePredict[i])
|
|
|
|
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
|
plt<span style="color: #666666">.</span>figure()
|
|
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'r'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
|
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
|
|
|
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
|
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
|
plt<span style="color: #666666">.</span>legend()
|
|
plt<span style="color: #666666">.</span>show()
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
|
We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.
|
|
What happens if we do not include the intercept in our fit?
|
|
Let us see how we can change this code by zero centering.
|
|
</p>
|
|
|
|
|
|
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
|
<div class="cell border-box-sizing code_cell rendered">
|
|
<div class="input">
|
|
<div class="inner_cell">
|
|
<div class="input_area">
|
|
<div class="highlight" style="background: #f8f8f8">
|
|
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
|
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
|
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
|
|
|
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
|
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
|
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
|
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
|
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
|
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
|
|
|
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
|
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
|
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
|
|
|
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
|
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
|
|
|
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
|
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
|
|
|
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
|
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
|
|
|
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
|
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
|
<span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
|
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
|
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
|
<span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)</span>
|
|
<span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
|
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
|
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler
|
|
|
|
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
|
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
|
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
|
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">6</span>
|
|
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
|
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
|
|
|
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">2</span>, nlambdas)
|
|
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
|
lmb <span style="color: #666666">=</span> lambdas[i]
|
|
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
|
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
|
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
|
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler
|
|
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
|
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
|
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
|
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
|
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
|
<span style="color: #008000">print</span>(intercept_)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
|
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for own Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(MSEOwnRidgePredict[i])
|
|
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
|
<span style="color: #008000">print</span>(MSERidgePredict[i])
|
|
|
|
|
|
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
|
plt<span style="color: #666666">.</span>figure()
|
|
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
|
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
|
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
|
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
|
plt<span style="color: #666666">.</span>legend()
|
|
plt<span style="color: #666666">.</span>show()
|
|
</pre>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
<div class="output_wrapper">
|
|
<div class="output">
|
|
<div class="output_area">
|
|
<div class="output_subarea output_stream output_stdout output_text">
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
</div>
|
|
|
|
<p>We see here, when compared to the code which includes explicitely the
|
|
intercept column, that our MSE value is actually smaller. This is
|
|
because the regularization term does not include the intercept value
|
|
\( \beta_0 \) in the fitting. This applies to Lasso regularization as
|
|
well. It means that our optimization is now done only with the
|
|
centered matrix and/or vector that enter the fitting procedure.
|
|
</p>
|
|
|
|
<p>
|
|
<!-- navigation buttons at the bottom of the page -->
|
|
<ul class="pagination">
|
|
<li><a href="._week35-bs061.html">«</a></li>
|
|
<li><a href="._week35-bs000.html">1</a></li>
|
|
<li><a href="">...</a></li>
|
|
<li><a href="._week35-bs054.html">55</a></li>
|
|
<li><a href="._week35-bs055.html">56</a></li>
|
|
<li><a href="._week35-bs056.html">57</a></li>
|
|
<li><a href="._week35-bs057.html">58</a></li>
|
|
<li><a href="._week35-bs058.html">59</a></li>
|
|
<li><a href="._week35-bs059.html">60</a></li>
|
|
<li><a href="._week35-bs060.html">61</a></li>
|
|
<li><a href="._week35-bs061.html">62</a></li>
|
|
<li class="active"><a href="._week35-bs062.html">63</a></li>
|
|
</ul>
|
|
<!-- ------------------- end of main content --------------- -->
|
|
</div> <!-- end container -->
|
|
<!-- include javascript, jQuery *first* -->
|
|
<script src="https://ajax.googleapis.com/ajax/libs/jquery/1.10.2/jquery.min.js"></script>
|
|
<script src="https://netdna.bootstrapcdn.com/bootstrap/3.0.0/js/bootstrap.min.js"></script>
|
|
<!-- Bootstrap footer
|
|
<footer>
|
|
<a href="https://..."><img width="250" align=right src="https://..."></a>
|
|
</footer>
|
|
-->
|
|
<center style="font-size:80%">
|
|
<!-- copyright only on the titlepage -->
|
|
</center>
|
|
</body>
|
|
</html>
|
|
|