update
This commit is contained in:
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
@@ -271,7 +276,8 @@ MathJax.Hub.Config({
|
||||
</ul>
|
||||
<li> Material for the lecture on Thursday September 7</li>
|
||||
<ul>
|
||||
<li> Linear Regression and links with Statistics, Resampling methods</li>
|
||||
<li> Technicalities related to scaling and other issues with data handling</li>
|
||||
<li> Linear Regression and links with Statistics</li>
|
||||
<li> Recommended Reading: Goodfellow et al chapter 3 on probability theory, see URL:""</li>
|
||||
<li> See also Murphy, sections 2.4 (Gaussian distributions) and 3.2 (Bayesian Statistics, basis)</li>
|
||||
</ul>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
@@ -261,6 +266,552 @@ MathJax.Hub.Config({
|
||||
<a name="part0029"></a>
|
||||
<!-- !split -->
|
||||
<h2 id="material-for-lecture-thursday-september-7" class="anchor">Material for lecture Thursday September 7 </h2>
|
||||
<h2 id="important-technicalities-more-on-rescaling-data" class="anchor">Important technicalities: More on Rescaling data </h2>
|
||||
|
||||
<p>When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some technicalities to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
</p>
|
||||
|
||||
<p>The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which tend confuse us.
|
||||
</p>
|
||||
|
||||
<p>The intercept can be interpreted as the expected value of our
|
||||
target/output variables when all other predictors are set to zero.
|
||||
Thus, if we cannot assume that the expected outputs/targets are zero
|
||||
when all predictors are zero (the columns in the design matrix), it
|
||||
may be a bad idea to implement a model which penalizes the intercept.
|
||||
Furthermore, in for example Ridge and Lasso regression, the default solutions
|
||||
from the library <b>Scikit-Learn</b> (when not shrinking \( \beta_0 \)) for the unknown parameters
|
||||
\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and
|
||||
\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values.
|
||||
</p>
|
||||
|
||||
<p>If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,
|
||||
the results may differ.
|
||||
</p>
|
||||
|
||||
<p>The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_self">Standardscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
</p>
|
||||
|
||||
<p>If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
penalization of the parameters since their magnitude depends on the
|
||||
scale of their corresponding predictor.
|
||||
</p>
|
||||
|
||||
<p>Suppose as an example that you
|
||||
you have an input variable given by the heights of different persons.
|
||||
Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters.
|
||||
This can clearly lead to problems in evaluating the cost/loss functions.
|
||||
</p>
|
||||
|
||||
<p>Keep in mind that when you transform your data set before training a model, the same transformation needs to be done
|
||||
on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #BA2121; font-style: italic">"""</span>
|
||||
<span style="color: #BA2121; font-style: italic">#Model training, we compute the mean value of y and X</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_train_mean = np.mean(y_train)</span>
|
||||
<span style="color: #BA2121; font-style: italic">X_train_mean = np.mean(X_train,axis=0)</span>
|
||||
<span style="color: #BA2121; font-style: italic">X_train = X_train - X_train_mean</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_train = y_train - y_train_mean</span>
|
||||
|
||||
<span style="color: #BA2121; font-style: italic"># The we fit our model with the training data</span>
|
||||
<span style="color: #BA2121; font-style: italic">trained_model = some_model.fit(X_train,y_train)</span>
|
||||
|
||||
|
||||
<span style="color: #BA2121; font-style: italic">#Model prediction, we need also to transform our data set used for the prediction.</span>
|
||||
<span style="color: #BA2121; font-style: italic">X_test = X_test - X_train_mean #Use mean from training data</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_pred = trained_model(X_test)</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_pred = y_pred + y_train_mean</span>
|
||||
<span style="color: #BA2121; font-style: italic">"""</span>
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>Let us try to understand what this may imply mathematically when we
|
||||
subtract the mean values, also known as <em>zero centering</em>. For
|
||||
simplicity, we will focus on ordinary regression, as done in the above example.
|
||||
</p>
|
||||
|
||||
<p>The cost/loss function for regression is</p>
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,.
|
||||
$$
|
||||
|
||||
<p>Recall also that we use the squared value. This expression can lead to an
|
||||
increased penalty for higher differences between predicted and
|
||||
output/target values.
|
||||
</p>
|
||||
|
||||
<p>What we have done is to single out the \( \beta_0 \) term in the
|
||||
definition of the mean squared error (MSE). The design matrix \( X \)
|
||||
does in this case not contain any intercept column. When we take the
|
||||
derivative with respect to \( \beta_0 \), we want the derivative to obey
|
||||
</p>
|
||||
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>for all \( j \). For \( \beta_0 \) we have</p>
|
||||
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right).
|
||||
$$
|
||||
|
||||
<p>Multiplying away the constant \( 2/n \), we obtain</p>
|
||||
$$
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
|
||||
<p>Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \).
|
||||
Our result for \( \beta_0 \) simplifies then to
|
||||
</p>
|
||||
$$
|
||||
n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1.
|
||||
$$
|
||||
|
||||
<p>We obtain then</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}.
|
||||
$$
|
||||
|
||||
<p>If we define</p>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1},
|
||||
$$
|
||||
|
||||
<p>and the mean value of the outputs as</p>
|
||||
$$
|
||||
\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i,
|
||||
$$
|
||||
|
||||
<p>we have</p>
|
||||
$$
|
||||
\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}.
|
||||
$$
|
||||
|
||||
<p>In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j.
|
||||
$$
|
||||
|
||||
<p>We can rewrite the latter equation as</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j,
|
||||
$$
|
||||
|
||||
<p>where we have defined</p>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
|
||||
<p>the mean value for all elements of the column vector \( \boldsymbol{x}_j \).</p>
|
||||
|
||||
<p>Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)</p>
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}).
|
||||
$$
|
||||
|
||||
<p>If we minimize with respect to \( \boldsymbol{\beta} \) we have then</p>
|
||||
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}},
|
||||
$$
|
||||
|
||||
<p>where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \).
|
||||
</p>
|
||||
|
||||
<p>For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then</p>
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}.
|
||||
$$
|
||||
|
||||
<p>What does this mean? And why do we insist on all this? Let us look at some examples.</p>
|
||||
|
||||
<p>This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
Note also that we do not split the data into training and test.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">2021</span>)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">fit_beta</span>(X, y):
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X) <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
||||
|
||||
|
||||
true_beta <span style="color: #666666">=</span> [<span style="color: #666666">2</span>, <span style="color: #666666">0.5</span>, <span style="color: #666666">3.7</span>]
|
||||
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>, <span style="color: #666666">1</span>, <span style="color: #666666">11</span>)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(
|
||||
np<span style="color: #666666">.</span>asarray([x <span style="color: #666666">**</span> p <span style="color: #666666">*</span> b <span style="color: #008000; font-weight: bold">for</span> p, b <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(true_beta)]), axis<span style="color: #666666">=0</span>
|
||||
) <span style="color: #666666">+</span> <span style="color: #666666">0.1</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(size<span style="color: #666666">=</span><span style="color: #008000">len</span>(x))
|
||||
|
||||
degree <span style="color: #666666">=</span> <span style="color: #666666">3</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree))
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> p
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is included in the design matrix</span>
|
||||
skl <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"True beta: </span><span style="color: #BB6688; font-weight: bold">{</span>true_beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
ypredictOwn <span style="color: #666666">=</span> X <span style="color: #666666">@</span> beta
|
||||
ypredictSKL <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>predict(X)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with intercept column"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictOwn))
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with intercept column from SKL"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>scatter(x, y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Data"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, skl<span style="color: #666666">.</span>predict(X), label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Do not include the intercept in the design matrix</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> (p <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is not included in the design matrix</span>
|
||||
skl <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(y, axis<span style="color: #666666">=0</span>)
|
||||
X_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(X, axis<span style="color: #666666">=0</span>)
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X <span style="color: #666666">-</span> X_offset, y <span style="color: #666666">-</span> y_offset)
|
||||
intercept <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_offset <span style="color: #666666">-</span> X_offset <span style="color: #666666">@</span> beta)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Manual intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>intercept<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta (wiothout intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>intercept_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta (without intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
ypredictOwn <span style="color: #666666">=</span> X <span style="color: #666666">@</span> beta
|
||||
ypredictSKL <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>predict(X)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with Manual intercept"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictOwn<span style="color: #666666">+</span>intercept))
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with Sklearn intercept"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta <span style="color: #666666">+</span> intercept, <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit (manual intercept)"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, skl<span style="color: #666666">.</span>predict(X), <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt<span style="color: #666666">.</span>grid()
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The intercept is the value of our output/target variable
|
||||
when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
|
||||
</p>
|
||||
|
||||
<p>Printing the MSE, we see first that both methods give the same MSE, as
|
||||
they should. However, when we move to for example Ridge regression,
|
||||
the way we treat the intercept may give a larger or smaller MSE,
|
||||
meaning that the MSE can be penalized by the value of the
|
||||
intercept. Not including the intercept in the fit, means that the
|
||||
regularization term does not include \( \beta_0 \). For different values
|
||||
of \( \lambda \), this may lead to different MSE values.
|
||||
</p>
|
||||
|
||||
<p>To remind the reader, the regularization term, with the intercept in Ridge regression, is given by</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2,
|
||||
$$
|
||||
|
||||
<p>but when we take out the intercept, this equation becomes</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2.
|
||||
$$
|
||||
|
||||
<p>For Lasso regression we have</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert.
|
||||
$$
|
||||
|
||||
<p>It means that, when scaling the design matrix and the outputs/targets,
|
||||
by subtracting the mean values, we have an optimization problem which
|
||||
is not penalized by the intercept. The MSE value can then be smaller
|
||||
since it focuses only on the remaining quantities. If we however bring
|
||||
back the intercept, we will get a MSE which then contains the
|
||||
intercept.
|
||||
</p>
|
||||
|
||||
<p>Armed with this wisdom, we attempt first to simply set the intercept equal to <b>False</b> in our implementation of Ridge regression for our well-known vanilla data set.</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree))
|
||||
<span style="color: #408080; font-style: italic">#We include explicitely the intercept column</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Maxpolydegree):
|
||||
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">6</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">2</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #408080; font-style: italic"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb,fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
ytildeOwnRidge <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ytildeRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSERidgePredict[i])
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'r'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.
|
||||
What happens if we do not include the intercept in our fit?
|
||||
Let us see how we can change this code by zero centering.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
<span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
<span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)</span>
|
||||
<span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">6</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">2</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #008000">print</span>(intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSERidgePredict[i])
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>We see here, when compared to the code which includes explicitely the
|
||||
intercept column, that our MSE value is actually smaller. This is
|
||||
because the regularization term does not include the intercept value
|
||||
\( \beta_0 \) in the fitting. This applies to Lasso regularization as
|
||||
well. It means that our optimization is now done only with the
|
||||
centered matrix and/or vector that enter the fitting procedure.
|
||||
</p>
|
||||
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -112,6 +112,10 @@ doconce format html week36.do.txt --html_style=bootstrap --pygments_html_style=d
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -229,6 +233,7 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs027.html#with-lasso-regression" style="font-size: 80%;">With Lasso Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs028.html#another-example-now-with-a-polynomial-fit" style="font-size: 80%;">Another Example, now with a polynomial fit</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#material-for-lecture-thursday-september-7" style="font-size: 80%;">Material for lecture Thursday September 7</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs029.html#important-technicalities-more-on-rescaling-data" style="font-size: 80%;">Important technicalities: More on Rescaling data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs030.html#linking-the-regression-analysis-with-a-statistical-interpretation" style="font-size: 80%;">Linking the regression analysis with a statistical interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs031.html#assumptions-made" style="font-size: 80%;">Assumptions made</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week36-bs032.html#expectation-value-and-variance" style="font-size: 80%;">Expectation value and variance</a></li>
|
||||
|
||||
@@ -211,7 +211,9 @@ MathJax.Hub.Config({
|
||||
<p><li> Material for the lecture on Thursday September 7</li>
|
||||
<ul>
|
||||
|
||||
<p><li> Linear Regression and links with Statistics, Resampling methods</li>
|
||||
<p><li> Technicalities related to scaling and other issues with data handling</li>
|
||||
|
||||
<p><li> Linear Regression and links with Statistics</li>
|
||||
|
||||
<p><li> Recommended Reading: Goodfellow et al chapter 3 on probability theory, see URL:""</li>
|
||||
|
||||
@@ -1293,6 +1295,588 @@ plt.show()
|
||||
|
||||
<section>
|
||||
<h2 id="material-for-lecture-thursday-september-7">Material for lecture Thursday September 7 </h2>
|
||||
<h2 id="important-technicalities-more-on-rescaling-data">Important technicalities: More on Rescaling data </h2>
|
||||
|
||||
<p>When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some technicalities to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
</p>
|
||||
|
||||
<p>The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which tend confuse us.
|
||||
</p>
|
||||
|
||||
<p>The intercept can be interpreted as the expected value of our
|
||||
target/output variables when all other predictors are set to zero.
|
||||
Thus, if we cannot assume that the expected outputs/targets are zero
|
||||
when all predictors are zero (the columns in the design matrix), it
|
||||
may be a bad idea to implement a model which penalizes the intercept.
|
||||
Furthermore, in for example Ridge and Lasso regression, the default solutions
|
||||
from the library <b>Scikit-Learn</b> (when not shrinking \( \beta_0 \)) for the unknown parameters
|
||||
\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and
|
||||
\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values.
|
||||
</p>
|
||||
|
||||
<p>If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,
|
||||
the results may differ.
|
||||
</p>
|
||||
|
||||
<p>The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_blank">Standardscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
</p>
|
||||
|
||||
<p>If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
penalization of the parameters since their magnitude depends on the
|
||||
scale of their corresponding predictor.
|
||||
</p>
|
||||
|
||||
<p>Suppose as an example that you
|
||||
you have an input variable given by the heights of different persons.
|
||||
Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters.
|
||||
This can clearly lead to problems in evaluating the cost/loss functions.
|
||||
</p>
|
||||
|
||||
<p>Keep in mind that when you transform your data set before training a model, the same transformation needs to be done
|
||||
on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="font-size: 80%; line-height: 125%;"><span style="color: #CD5555">"""</span>
|
||||
<span style="color: #CD5555">#Model training, we compute the mean value of y and X</span>
|
||||
<span style="color: #CD5555">y_train_mean = np.mean(y_train)</span>
|
||||
<span style="color: #CD5555">X_train_mean = np.mean(X_train,axis=0)</span>
|
||||
<span style="color: #CD5555">X_train = X_train - X_train_mean</span>
|
||||
<span style="color: #CD5555">y_train = y_train - y_train_mean</span>
|
||||
|
||||
<span style="color: #CD5555"># The we fit our model with the training data</span>
|
||||
<span style="color: #CD5555">trained_model = some_model.fit(X_train,y_train)</span>
|
||||
|
||||
|
||||
<span style="color: #CD5555">#Model prediction, we need also to transform our data set used for the prediction.</span>
|
||||
<span style="color: #CD5555">X_test = X_test - X_train_mean #Use mean from training data</span>
|
||||
<span style="color: #CD5555">y_pred = trained_model(X_test)</span>
|
||||
<span style="color: #CD5555">y_pred = y_pred + y_train_mean</span>
|
||||
<span style="color: #CD5555">"""</span>
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>Let us try to understand what this may imply mathematically when we
|
||||
subtract the mean values, also known as <em>zero centering</em>. For
|
||||
simplicity, we will focus on ordinary regression, as done in the above example.
|
||||
</p>
|
||||
|
||||
<p>The cost/loss function for regression is</p>
|
||||
<p> <br>
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>Recall also that we use the squared value. This expression can lead to an
|
||||
increased penalty for higher differences between predicted and
|
||||
output/target values.
|
||||
</p>
|
||||
|
||||
<p>What we have done is to single out the \( \beta_0 \) term in the
|
||||
definition of the mean squared error (MSE). The design matrix \( X \)
|
||||
does in this case not contain any intercept column. When we take the
|
||||
derivative with respect to \( \beta_0 \), we want the derivative to obey
|
||||
</p>
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>for all \( j \). For \( \beta_0 \) we have</p>
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right).
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>Multiplying away the constant \( 2/n \), we obtain</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \).
|
||||
Our result for \( \beta_0 \) simplifies then to
|
||||
</p>
|
||||
<p> <br>
|
||||
$$
|
||||
n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>We obtain then</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>If we define</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>and the mean value of the outputs as</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>we have</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>We can rewrite the latter equation as</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>where we have defined</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>the mean value for all elements of the column vector \( \boldsymbol{x}_j \).</p>
|
||||
|
||||
<p>Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)</p>
|
||||
<p> <br>
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}).
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>If we minimize with respect to \( \boldsymbol{\beta} \) we have then</p>
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \).
|
||||
</p>
|
||||
|
||||
<p>For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>What does this mean? And why do we insist on all this? Let us look at some examples.</p>
|
||||
|
||||
<p>This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
Note also that we do not split the data into training and test.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="font-size: 80%; line-height: 125%;"><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np.random.seed(<span style="color: #B452CD">2021</span>)
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">fit_beta</span>(X, y):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.linalg.pinv(X.T @ X) @ X.T @ y
|
||||
|
||||
|
||||
true_beta = [<span style="color: #B452CD">2</span>, <span style="color: #B452CD">0.5</span>, <span style="color: #B452CD">3.7</span>]
|
||||
|
||||
x = np.linspace(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">1</span>, <span style="color: #B452CD">11</span>)
|
||||
y = np.sum(
|
||||
np.asarray([x ** p * b <span style="color: #8B008B; font-weight: bold">for</span> p, b <span style="color: #8B008B">in</span> <span style="color: #658b00">enumerate</span>(true_beta)]), axis=<span style="color: #B452CD">0</span>
|
||||
) + <span style="color: #B452CD">0.1</span> * np.random.normal(size=<span style="color: #658b00">len</span>(x))
|
||||
|
||||
degree = <span style="color: #B452CD">3</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree))
|
||||
|
||||
<span style="color: #228B22"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree):
|
||||
X[:, p] = x ** p
|
||||
|
||||
beta = fit_beta(X, y)
|
||||
|
||||
<span style="color: #228B22"># Intercept is included in the design matrix</span>
|
||||
skl = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>).fit(X, y)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"True beta: {</span>true_beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta: {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta: {</span>skl.coef_<span style="color: #CD5555">}"</span>)
|
||||
ypredictOwn = X @ beta
|
||||
ypredictSKL = skl.predict(X)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with intercept column"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictOwn))
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with intercept column from SKL"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
|
||||
plt.figure()
|
||||
plt.scatter(x, y, label=<span style="color: #CD5555">"Data"</span>)
|
||||
plt.plot(x, X @ beta, label=<span style="color: #CD5555">"Fit"</span>)
|
||||
plt.plot(x, skl.predict(X), label=<span style="color: #CD5555">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #228B22"># Do not include the intercept in the design matrix</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree - <span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree - <span style="color: #B452CD">1</span>):
|
||||
X[:, p] = x ** (p + <span style="color: #B452CD">1</span>)
|
||||
|
||||
<span style="color: #228B22"># Intercept is not included in the design matrix</span>
|
||||
skl = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">True</span>).fit(X, y)
|
||||
|
||||
<span style="color: #228B22"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset = np.average(y, axis=<span style="color: #B452CD">0</span>)
|
||||
X_offset = np.average(X, axis=<span style="color: #B452CD">0</span>)
|
||||
|
||||
beta = fit_beta(X - X_offset, y - y_offset)
|
||||
intercept = np.mean(y_offset - X_offset @ beta)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Manual intercept: {</span>intercept<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta (wiothout intercept): {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn intercept: {</span>skl.intercept_<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta (without intercept): {</span>skl.coef_<span style="color: #CD5555">}"</span>)
|
||||
ypredictOwn = X @ beta
|
||||
ypredictSKL = skl.predict(X)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with Manual intercept"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictOwn+intercept))
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with Sklearn intercept"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
plt.plot(x, X @ beta + intercept, <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Fit (manual intercept)"</span>)
|
||||
plt.plot(x, skl.predict(X), <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt.grid()
|
||||
plt.legend()
|
||||
|
||||
plt.show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The intercept is the value of our output/target variable
|
||||
when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
|
||||
</p>
|
||||
|
||||
<p>Printing the MSE, we see first that both methods give the same MSE, as
|
||||
they should. However, when we move to for example Ridge regression,
|
||||
the way we treat the intercept may give a larger or smaller MSE,
|
||||
meaning that the MSE can be penalized by the value of the
|
||||
intercept. Not including the intercept in the fit, means that the
|
||||
regularization term does not include \( \beta_0 \). For different values
|
||||
of \( \lambda \), this may lead to different MSE values.
|
||||
</p>
|
||||
|
||||
<p>To remind the reader, the regularization term, with the intercept in Ridge regression, is given by</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>but when we take out the intercept, this equation becomes</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>For Lasso regression we have</p>
|
||||
<p> <br>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>It means that, when scaling the design matrix and the outputs/targets,
|
||||
by subtracting the mean values, we have an optimization problem which
|
||||
is not penalized by the intercept. The MSE value can then be smaller
|
||||
since it focuses only on the remaining quantities. If we however bring
|
||||
back the intercept, we will get a MSE which then contains the
|
||||
intercept.
|
||||
</p>
|
||||
|
||||
<p>Armed with this wisdom, we attempt first to simply set the intercept equal to <b>False</b> in our implementation of Ridge regression for our well-known vanilla data set.</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="font-size: 80%; line-height: 125%;"><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">pandas</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pd</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">3155</span>)
|
||||
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree))
|
||||
<span style="color: #228B22">#We include explicitely the intercept column</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(Maxpolydegree):
|
||||
X[:,degree] = x**degree
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
p = Maxpolydegree
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">6</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">2</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
|
||||
<span style="color: #228B22"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge = linear_model.Ridge(lmb,fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
<span style="color: #228B22"># and then make the prediction</span>
|
||||
ytildeOwnRidge = X_train @ OwnRidgeBeta
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta
|
||||
ytildeRidge = RegRidge.predict(X_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSERidgePredict[i])
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'r'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g'</span>, label = <span style="color: #CD5555">'MSE Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.
|
||||
What happens if we do not include the intercept in our fit?
|
||||
Let us see how we can change this code by zero centering.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="font-size: 80%; line-height: 125%;"><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">pandas</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pd</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">315</span>)
|
||||
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree-<span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,Maxpolydegree): <span style="color: #228B22">#No intercept column</span>
|
||||
X[:,degree-<span style="color: #B452CD">1</span>] = x**(degree)
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
<span style="color: #228B22">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
<span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
<span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)</span>
|
||||
<span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
p = Maxpolydegree-<span style="color: #B452CD">1</span>
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">6</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">2</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean<span style="color: #707a7c">@OwnRidgeBeta</span> <span style="color: #228B22">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
<span style="color: #228B22">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta) <span style="color: #228B22">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #658b00">print</span>(intercept_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.intercept_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSERidgePredict[i])
|
||||
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'b--'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g--'</span>, label = <span style="color: #CD5555">'MSE SL Ridge Test'</span>)
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>We see here, when compared to the code which includes explicitely the
|
||||
intercept column, that our MSE value is actually smaller. This is
|
||||
because the regularization term does not include the intercept value
|
||||
\( \beta_0 \) in the fitting. This applies to Lasso regularization as
|
||||
well. It means that our optimization is now done only with the
|
||||
centered matrix and/or vector that enter the fitting procedure.
|
||||
</p>
|
||||
</section>
|
||||
|
||||
<section>
|
||||
|
||||
@@ -139,6 +139,10 @@ div.toc p,a {
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -246,7 +250,8 @@ MathJax.Hub.Config({
|
||||
</ul>
|
||||
<li> Material for the lecture on Thursday September 7</li>
|
||||
<ul>
|
||||
<li> Linear Regression and links with Statistics, Resampling methods</li>
|
||||
<li> Technicalities related to scaling and other issues with data handling</li>
|
||||
<li> Linear Regression and links with Statistics</li>
|
||||
<li> Recommended Reading: Goodfellow et al chapter 3 on probability theory, see URL:""</li>
|
||||
<li> See also Murphy, sections 2.4 (Gaussian distributions) and 3.2 (Bayesian Statistics, basis)</li>
|
||||
</ul>
|
||||
@@ -1187,6 +1192,552 @@ plt.show()
|
||||
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
<h2 id="material-for-lecture-thursday-september-7">Material for lecture Thursday September 7 </h2>
|
||||
<h2 id="important-technicalities-more-on-rescaling-data">Important technicalities: More on Rescaling data </h2>
|
||||
|
||||
<p>When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some technicalities to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
</p>
|
||||
|
||||
<p>The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which tend confuse us.
|
||||
</p>
|
||||
|
||||
<p>The intercept can be interpreted as the expected value of our
|
||||
target/output variables when all other predictors are set to zero.
|
||||
Thus, if we cannot assume that the expected outputs/targets are zero
|
||||
when all predictors are zero (the columns in the design matrix), it
|
||||
may be a bad idea to implement a model which penalizes the intercept.
|
||||
Furthermore, in for example Ridge and Lasso regression, the default solutions
|
||||
from the library <b>Scikit-Learn</b> (when not shrinking \( \beta_0 \)) for the unknown parameters
|
||||
\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and
|
||||
\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values.
|
||||
</p>
|
||||
|
||||
<p>If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,
|
||||
the results may differ.
|
||||
</p>
|
||||
|
||||
<p>The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_blank">Standardscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
</p>
|
||||
|
||||
<p>If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
penalization of the parameters since their magnitude depends on the
|
||||
scale of their corresponding predictor.
|
||||
</p>
|
||||
|
||||
<p>Suppose as an example that you
|
||||
you have an input variable given by the heights of different persons.
|
||||
Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters.
|
||||
This can clearly lead to problems in evaluating the cost/loss functions.
|
||||
</p>
|
||||
|
||||
<p>Keep in mind that when you transform your data set before training a model, the same transformation needs to be done
|
||||
on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="line-height: 125%;"><span style="color: #CD5555">"""</span>
|
||||
<span style="color: #CD5555">#Model training, we compute the mean value of y and X</span>
|
||||
<span style="color: #CD5555">y_train_mean = np.mean(y_train)</span>
|
||||
<span style="color: #CD5555">X_train_mean = np.mean(X_train,axis=0)</span>
|
||||
<span style="color: #CD5555">X_train = X_train - X_train_mean</span>
|
||||
<span style="color: #CD5555">y_train = y_train - y_train_mean</span>
|
||||
|
||||
<span style="color: #CD5555"># The we fit our model with the training data</span>
|
||||
<span style="color: #CD5555">trained_model = some_model.fit(X_train,y_train)</span>
|
||||
|
||||
|
||||
<span style="color: #CD5555">#Model prediction, we need also to transform our data set used for the prediction.</span>
|
||||
<span style="color: #CD5555">X_test = X_test - X_train_mean #Use mean from training data</span>
|
||||
<span style="color: #CD5555">y_pred = trained_model(X_test)</span>
|
||||
<span style="color: #CD5555">y_pred = y_pred + y_train_mean</span>
|
||||
<span style="color: #CD5555">"""</span>
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>Let us try to understand what this may imply mathematically when we
|
||||
subtract the mean values, also known as <em>zero centering</em>. For
|
||||
simplicity, we will focus on ordinary regression, as done in the above example.
|
||||
</p>
|
||||
|
||||
<p>The cost/loss function for regression is</p>
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,.
|
||||
$$
|
||||
|
||||
<p>Recall also that we use the squared value. This expression can lead to an
|
||||
increased penalty for higher differences between predicted and
|
||||
output/target values.
|
||||
</p>
|
||||
|
||||
<p>What we have done is to single out the \( \beta_0 \) term in the
|
||||
definition of the mean squared error (MSE). The design matrix \( X \)
|
||||
does in this case not contain any intercept column. When we take the
|
||||
derivative with respect to \( \beta_0 \), we want the derivative to obey
|
||||
</p>
|
||||
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>for all \( j \). For \( \beta_0 \) we have</p>
|
||||
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right).
|
||||
$$
|
||||
|
||||
<p>Multiplying away the constant \( 2/n \), we obtain</p>
|
||||
$$
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
|
||||
<p>Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \).
|
||||
Our result for \( \beta_0 \) simplifies then to
|
||||
</p>
|
||||
$$
|
||||
n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1.
|
||||
$$
|
||||
|
||||
<p>We obtain then</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}.
|
||||
$$
|
||||
|
||||
<p>If we define</p>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1},
|
||||
$$
|
||||
|
||||
<p>and the mean value of the outputs as</p>
|
||||
$$
|
||||
\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i,
|
||||
$$
|
||||
|
||||
<p>we have</p>
|
||||
$$
|
||||
\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}.
|
||||
$$
|
||||
|
||||
<p>In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j.
|
||||
$$
|
||||
|
||||
<p>We can rewrite the latter equation as</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j,
|
||||
$$
|
||||
|
||||
<p>where we have defined</p>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
|
||||
<p>the mean value for all elements of the column vector \( \boldsymbol{x}_j \).</p>
|
||||
|
||||
<p>Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)</p>
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}).
|
||||
$$
|
||||
|
||||
<p>If we minimize with respect to \( \boldsymbol{\beta} \) we have then</p>
|
||||
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}},
|
||||
$$
|
||||
|
||||
<p>where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \).
|
||||
</p>
|
||||
|
||||
<p>For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then</p>
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}.
|
||||
$$
|
||||
|
||||
<p>What does this mean? And why do we insist on all this? Let us look at some examples.</p>
|
||||
|
||||
<p>This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
Note also that we do not split the data into training and test.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="line-height: 125%;"><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.linear_model</span> <span style="color: #8B008B; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np.random.seed(<span style="color: #B452CD">2021</span>)
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">fit_beta</span>(X, y):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.linalg.pinv(X.T @ X) @ X.T @ y
|
||||
|
||||
|
||||
true_beta = [<span style="color: #B452CD">2</span>, <span style="color: #B452CD">0.5</span>, <span style="color: #B452CD">3.7</span>]
|
||||
|
||||
x = np.linspace(<span style="color: #B452CD">0</span>, <span style="color: #B452CD">1</span>, <span style="color: #B452CD">11</span>)
|
||||
y = np.sum(
|
||||
np.asarray([x ** p * b <span style="color: #8B008B; font-weight: bold">for</span> p, b <span style="color: #8B008B">in</span> <span style="color: #658b00">enumerate</span>(true_beta)]), axis=<span style="color: #B452CD">0</span>
|
||||
) + <span style="color: #B452CD">0.1</span> * np.random.normal(size=<span style="color: #658b00">len</span>(x))
|
||||
|
||||
degree = <span style="color: #B452CD">3</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree))
|
||||
|
||||
<span style="color: #228B22"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree):
|
||||
X[:, p] = x ** p
|
||||
|
||||
beta = fit_beta(X, y)
|
||||
|
||||
<span style="color: #228B22"># Intercept is included in the design matrix</span>
|
||||
skl = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>).fit(X, y)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"True beta: {</span>true_beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta: {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta: {</span>skl.coef_<span style="color: #CD5555">}"</span>)
|
||||
ypredictOwn = X @ beta
|
||||
ypredictSKL = skl.predict(X)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with intercept column"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictOwn))
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with intercept column from SKL"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
|
||||
plt.figure()
|
||||
plt.scatter(x, y, label=<span style="color: #CD5555">"Data"</span>)
|
||||
plt.plot(x, X @ beta, label=<span style="color: #CD5555">"Fit"</span>)
|
||||
plt.plot(x, skl.predict(X), label=<span style="color: #CD5555">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #228B22"># Do not include the intercept in the design matrix</span>
|
||||
X = np.zeros((<span style="color: #658b00">len</span>(x), degree - <span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> p <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(degree - <span style="color: #B452CD">1</span>):
|
||||
X[:, p] = x ** (p + <span style="color: #B452CD">1</span>)
|
||||
|
||||
<span style="color: #228B22"># Intercept is not included in the design matrix</span>
|
||||
skl = LinearRegression(fit_intercept=<span style="color: #8B008B; font-weight: bold">True</span>).fit(X, y)
|
||||
|
||||
<span style="color: #228B22"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset = np.average(y, axis=<span style="color: #B452CD">0</span>)
|
||||
X_offset = np.average(X, axis=<span style="color: #B452CD">0</span>)
|
||||
|
||||
beta = fit_beta(X - X_offset, y - y_offset)
|
||||
intercept = np.mean(y_offset - X_offset @ beta)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Manual intercept: {</span>intercept<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Fitted beta (wiothout intercept): {</span>beta<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn intercept: {</span>skl.intercept_<span style="color: #CD5555">}"</span>)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"Sklearn fitted beta (without intercept): {</span>skl.coef_<span style="color: #CD5555">}"</span>)
|
||||
ypredictOwn = X @ beta
|
||||
ypredictSKL = skl.predict(X)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with Manual intercept"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictOwn+intercept))
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">f"MSE with Sklearn intercept"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
plt.plot(x, X @ beta + intercept, <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Fit (manual intercept)"</span>)
|
||||
plt.plot(x, skl.predict(X), <span style="color: #CD5555">"--"</span>, label=<span style="color: #CD5555">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt.grid()
|
||||
plt.legend()
|
||||
|
||||
plt.show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The intercept is the value of our output/target variable
|
||||
when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
|
||||
</p>
|
||||
|
||||
<p>Printing the MSE, we see first that both methods give the same MSE, as
|
||||
they should. However, when we move to for example Ridge regression,
|
||||
the way we treat the intercept may give a larger or smaller MSE,
|
||||
meaning that the MSE can be penalized by the value of the
|
||||
intercept. Not including the intercept in the fit, means that the
|
||||
regularization term does not include \( \beta_0 \). For different values
|
||||
of \( \lambda \), this may lead to different MSE values.
|
||||
</p>
|
||||
|
||||
<p>To remind the reader, the regularization term, with the intercept in Ridge regression, is given by</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2,
|
||||
$$
|
||||
|
||||
<p>but when we take out the intercept, this equation becomes</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2.
|
||||
$$
|
||||
|
||||
<p>For Lasso regression we have</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert.
|
||||
$$
|
||||
|
||||
<p>It means that, when scaling the design matrix and the outputs/targets,
|
||||
by subtracting the mean values, we have an optimization problem which
|
||||
is not penalized by the intercept. The MSE value can then be smaller
|
||||
since it focuses only on the remaining quantities. If we however bring
|
||||
back the intercept, we will get a MSE which then contains the
|
||||
intercept.
|
||||
</p>
|
||||
|
||||
<p>Armed with this wisdom, we attempt first to simply set the intercept equal to <b>False</b> in our implementation of Ridge regression for our well-known vanilla data set.</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="line-height: 125%;"><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">pandas</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pd</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">3155</span>)
|
||||
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree))
|
||||
<span style="color: #228B22">#We include explicitely the intercept column</span>
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(Maxpolydegree):
|
||||
X[:,degree] = x**degree
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
p = Maxpolydegree
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">6</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">2</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
|
||||
<span style="color: #228B22"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge = linear_model.Ridge(lmb,fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
<span style="color: #228B22"># and then make the prediction</span>
|
||||
ytildeOwnRidge = X_train @ OwnRidgeBeta
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta
|
||||
ytildeRidge = RegRidge.predict(X_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSERidgePredict[i])
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'r'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g'</span>, label = <span style="color: #CD5555">'MSE Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.
|
||||
What happens if we do not include the intercept in our fit?
|
||||
Let us see how we can change this code by zero centering.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #eeeedd">
|
||||
<pre style="line-height: 125%;"><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">pandas</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">pd</span>
|
||||
<span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">matplotlib.pyplot</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">plt</span>
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">315</span>)
|
||||
|
||||
n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree-<span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,Maxpolydegree): <span style="color: #228B22">#No intercept column</span>
|
||||
X[:,degree-<span style="color: #B452CD">1</span>] = x**(degree)
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
<span style="color: #228B22">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
<span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
<span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)</span>
|
||||
<span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
p = Maxpolydegree-<span style="color: #B452CD">1</span>
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">6</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">2</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean<span style="color: #707a7c">@OwnRidgeBeta</span> <span style="color: #228B22">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
<span style="color: #228B22">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta) <span style="color: #228B22">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #658b00">print</span>(intercept_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.intercept_)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(MSERidgePredict[i])
|
||||
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'b--'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g--'</span>, label = <span style="color: #CD5555">'MSE SL Ridge Test'</span>)
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
plt.show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>We see here, when compared to the code which includes explicitely the
|
||||
intercept column, that our MSE value is actually smaller. This is
|
||||
because the regularization term does not include the intercept value
|
||||
\( \beta_0 \) in the fitting. This applies to Lasso regularization as
|
||||
well. It means that our optimization is now done only with the
|
||||
centered matrix and/or vector that enter the fitting procedure.
|
||||
</p>
|
||||
|
||||
<!-- !split -->
|
||||
<h2 id="linking-the-regression-analysis-with-a-statistical-interpretation">Linking the regression analysis with a statistical interpretation </h2>
|
||||
|
||||
@@ -216,6 +216,10 @@ div.toc p,a {
|
||||
2,
|
||||
None,
|
||||
'material-for-lecture-thursday-september-7'),
|
||||
('Important technicalities: More on Rescaling data',
|
||||
2,
|
||||
None,
|
||||
'important-technicalities-more-on-rescaling-data'),
|
||||
('Linking the regression analysis with a statistical '
|
||||
'interpretation',
|
||||
2,
|
||||
@@ -323,7 +327,8 @@ MathJax.Hub.Config({
|
||||
</ul>
|
||||
<li> Material for the lecture on Thursday September 7</li>
|
||||
<ul>
|
||||
<li> Linear Regression and links with Statistics, Resampling methods</li>
|
||||
<li> Technicalities related to scaling and other issues with data handling</li>
|
||||
<li> Linear Regression and links with Statistics</li>
|
||||
<li> Recommended Reading: Goodfellow et al chapter 3 on probability theory, see URL:""</li>
|
||||
<li> See also Murphy, sections 2.4 (Gaussian distributions) and 3.2 (Bayesian Statistics, basis)</li>
|
||||
</ul>
|
||||
@@ -1264,6 +1269,552 @@ plt<span style="color: #666666">.</span>show()
|
||||
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
<h2 id="material-for-lecture-thursday-september-7">Material for lecture Thursday September 7 </h2>
|
||||
<h2 id="important-technicalities-more-on-rescaling-data">Important technicalities: More on Rescaling data </h2>
|
||||
|
||||
<p>When you are comparing your own code with for example <b>Scikit-Learn</b>'s
|
||||
library, there are some technicalities to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
</p>
|
||||
|
||||
<p>The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which tend confuse us.
|
||||
</p>
|
||||
|
||||
<p>The intercept can be interpreted as the expected value of our
|
||||
target/output variables when all other predictors are set to zero.
|
||||
Thus, if we cannot assume that the expected outputs/targets are zero
|
||||
when all predictors are zero (the columns in the design matrix), it
|
||||
may be a bad idea to implement a model which penalizes the intercept.
|
||||
Furthermore, in for example Ridge and Lasso regression, the default solutions
|
||||
from the library <b>Scikit-Learn</b> (when not shrinking \( \beta_0 \)) for the unknown parameters
|
||||
\( \boldsymbol{\beta} \), are derived under the assumption that both \( \boldsymbol{y} \) and
|
||||
\( \boldsymbol{X} \) are zero centered, that is we subtract the mean values.
|
||||
</p>
|
||||
|
||||
<p>If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,
|
||||
the results may differ.
|
||||
</p>
|
||||
|
||||
<p>The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_blank">Standardscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
</p>
|
||||
|
||||
<p>If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
penalization of the parameters since their magnitude depends on the
|
||||
scale of their corresponding predictor.
|
||||
</p>
|
||||
|
||||
<p>Suppose as an example that you
|
||||
you have an input variable given by the heights of different persons.
|
||||
Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters.
|
||||
This can clearly lead to problems in evaluating the cost/loss functions.
|
||||
</p>
|
||||
|
||||
<p>Keep in mind that when you transform your data set before training a model, the same transformation needs to be done
|
||||
on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #BA2121; font-style: italic">"""</span>
|
||||
<span style="color: #BA2121; font-style: italic">#Model training, we compute the mean value of y and X</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_train_mean = np.mean(y_train)</span>
|
||||
<span style="color: #BA2121; font-style: italic">X_train_mean = np.mean(X_train,axis=0)</span>
|
||||
<span style="color: #BA2121; font-style: italic">X_train = X_train - X_train_mean</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_train = y_train - y_train_mean</span>
|
||||
|
||||
<span style="color: #BA2121; font-style: italic"># The we fit our model with the training data</span>
|
||||
<span style="color: #BA2121; font-style: italic">trained_model = some_model.fit(X_train,y_train)</span>
|
||||
|
||||
|
||||
<span style="color: #BA2121; font-style: italic">#Model prediction, we need also to transform our data set used for the prediction.</span>
|
||||
<span style="color: #BA2121; font-style: italic">X_test = X_test - X_train_mean #Use mean from training data</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_pred = trained_model(X_test)</span>
|
||||
<span style="color: #BA2121; font-style: italic">y_pred = y_pred + y_train_mean</span>
|
||||
<span style="color: #BA2121; font-style: italic">"""</span>
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>Let us try to understand what this may imply mathematically when we
|
||||
subtract the mean values, also known as <em>zero centering</em>. For
|
||||
simplicity, we will focus on ordinary regression, as done in the above example.
|
||||
</p>
|
||||
|
||||
<p>The cost/loss function for regression is</p>
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,.
|
||||
$$
|
||||
|
||||
<p>Recall also that we use the squared value. This expression can lead to an
|
||||
increased penalty for higher differences between predicted and
|
||||
output/target values.
|
||||
</p>
|
||||
|
||||
<p>What we have done is to single out the \( \beta_0 \) term in the
|
||||
definition of the mean squared error (MSE). The design matrix \( X \)
|
||||
does in this case not contain any intercept column. When we take the
|
||||
derivative with respect to \( \beta_0 \), we want the derivative to obey
|
||||
</p>
|
||||
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>for all \( j \). For \( \beta_0 \) we have</p>
|
||||
|
||||
$$
|
||||
\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right).
|
||||
$$
|
||||
|
||||
<p>Multiplying away the constant \( 2/n \), we obtain</p>
|
||||
$$
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
|
||||
<p>Let us specialize first to the case where we have only two parameters \( \beta_0 \) and \( \beta_1 \).
|
||||
Our result for \( \beta_0 \) simplifies then to
|
||||
</p>
|
||||
$$
|
||||
n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1.
|
||||
$$
|
||||
|
||||
<p>We obtain then</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}.
|
||||
$$
|
||||
|
||||
<p>If we define</p>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1},
|
||||
$$
|
||||
|
||||
<p>and the mean value of the outputs as</p>
|
||||
$$
|
||||
\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i,
|
||||
$$
|
||||
|
||||
<p>we have</p>
|
||||
$$
|
||||
\beta_0 = \mu_y - \beta_1\mu_{\boldsymbol{x}_1}.
|
||||
$$
|
||||
|
||||
<p>In the general case with more parameters than \( \beta_0 \) and \( \beta_1 \), we have</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j.
|
||||
$$
|
||||
|
||||
<p>We can rewrite the latter equation as</p>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\boldsymbol{x}_j}\beta_j,
|
||||
$$
|
||||
|
||||
<p>where we have defined</p>
|
||||
$$
|
||||
\mu_{\boldsymbol{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
|
||||
<p>the mean value for all elements of the column vector \( \boldsymbol{x}_j \).</p>
|
||||
|
||||
<p>Replacing \( y_i \) with \( y_i - y_i - \overline{\boldsymbol{y}} \) and centering also our design matrix results in a cost function (in vector-matrix disguise)</p>
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}).
|
||||
$$
|
||||
|
||||
<p>If we minimize with respect to \( \boldsymbol{\beta} \) we have then</p>
|
||||
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}},
|
||||
$$
|
||||
|
||||
<p>where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj} \).
|
||||
</p>
|
||||
|
||||
<p>For Ridge regression we need to add \( \lambda \boldsymbol{\beta}^T\boldsymbol{\beta} \) to the cost function and get then</p>
|
||||
$$
|
||||
\hat{\boldsymbol{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}.
|
||||
$$
|
||||
|
||||
<p>What does this mean? And why do we insist on all this? Let us look at some examples.</p>
|
||||
|
||||
<p>This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
Note also that we do not split the data into training and test.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression
|
||||
|
||||
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">2021</span>)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">fit_beta</span>(X, y):
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X) <span style="color: #666666">@</span> X<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
||||
|
||||
|
||||
true_beta <span style="color: #666666">=</span> [<span style="color: #666666">2</span>, <span style="color: #666666">0.5</span>, <span style="color: #666666">3.7</span>]
|
||||
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linspace(<span style="color: #666666">0</span>, <span style="color: #666666">1</span>, <span style="color: #666666">11</span>)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>sum(
|
||||
np<span style="color: #666666">.</span>asarray([x <span style="color: #666666">**</span> p <span style="color: #666666">*</span> b <span style="color: #008000; font-weight: bold">for</span> p, b <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(true_beta)]), axis<span style="color: #666666">=0</span>
|
||||
) <span style="color: #666666">+</span> <span style="color: #666666">0.1</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>normal(size<span style="color: #666666">=</span><span style="color: #008000">len</span>(x))
|
||||
|
||||
degree <span style="color: #666666">=</span> <span style="color: #666666">3</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree))
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Include the intercept in the design matrix</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> p
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is included in the design matrix</span>
|
||||
skl <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"True beta: </span><span style="color: #BB6688; font-weight: bold">{</span>true_beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta: </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
ypredictOwn <span style="color: #666666">=</span> X <span style="color: #666666">@</span> beta
|
||||
ypredictSKL <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>predict(X)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with intercept column"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictOwn))
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with intercept column from SKL"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>scatter(x, y, label<span style="color: #666666">=</span><span style="color: #BA2121">"Data"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, skl<span style="color: #666666">.</span>predict(X), label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=False)"</span>)
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Do not include the intercept in the design matrix</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((<span style="color: #008000">len</span>(x), degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> p <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(degree <span style="color: #666666">-</span> <span style="color: #666666">1</span>):
|
||||
X[:, p] <span style="color: #666666">=</span> x <span style="color: #666666">**</span> (p <span style="color: #666666">+</span> <span style="color: #666666">1</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Intercept is not included in the design matrix</span>
|
||||
skl <span style="color: #666666">=</span> LinearRegression(fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)<span style="color: #666666">.</span>fit(X, y)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Use centered values for X and y when computing coefficients</span>
|
||||
y_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(y, axis<span style="color: #666666">=0</span>)
|
||||
X_offset <span style="color: #666666">=</span> np<span style="color: #666666">.</span>average(X, axis<span style="color: #666666">=0</span>)
|
||||
|
||||
beta <span style="color: #666666">=</span> fit_beta(X <span style="color: #666666">-</span> X_offset, y <span style="color: #666666">-</span> y_offset)
|
||||
intercept <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_offset <span style="color: #666666">-</span> X_offset <span style="color: #666666">@</span> beta)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Manual intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>intercept<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Fitted beta (wiothout intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>beta<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn intercept: </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>intercept_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"Sklearn fitted beta (without intercept): </span><span style="color: #BB6688; font-weight: bold">{</span>skl<span style="color: #666666">.</span>coef_<span style="color: #BB6688; font-weight: bold">}</span><span style="color: #BA2121">"</span>)
|
||||
ypredictOwn <span style="color: #666666">=</span> X <span style="color: #666666">@</span> beta
|
||||
ypredictSKL <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>predict(X)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with Manual intercept"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictOwn<span style="color: #666666">+</span>intercept))
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">f"MSE with Sklearn intercept"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y,ypredictSKL))
|
||||
|
||||
plt<span style="color: #666666">.</span>plot(x, X <span style="color: #666666">@</span> beta <span style="color: #666666">+</span> intercept, <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Fit (manual intercept)"</span>)
|
||||
plt<span style="color: #666666">.</span>plot(x, skl<span style="color: #666666">.</span>predict(X), <span style="color: #BA2121">"--"</span>, label<span style="color: #666666">=</span><span style="color: #BA2121">"Sklearn (fit_intercept=True)"</span>)
|
||||
plt<span style="color: #666666">.</span>grid()
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The intercept is the value of our output/target variable
|
||||
when all our features are zero and our function crosses the \( y \)-axis (for a one-dimensional case).
|
||||
</p>
|
||||
|
||||
<p>Printing the MSE, we see first that both methods give the same MSE, as
|
||||
they should. However, when we move to for example Ridge regression,
|
||||
the way we treat the intercept may give a larger or smaller MSE,
|
||||
meaning that the MSE can be penalized by the value of the
|
||||
intercept. Not including the intercept in the fit, means that the
|
||||
regularization term does not include \( \beta_0 \). For different values
|
||||
of \( \lambda \), this may lead to different MSE values.
|
||||
</p>
|
||||
|
||||
<p>To remind the reader, the regularization term, with the intercept in Ridge regression, is given by</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2,
|
||||
$$
|
||||
|
||||
<p>but when we take out the intercept, this equation becomes</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2.
|
||||
$$
|
||||
|
||||
<p>For Lasso regression we have</p>
|
||||
$$
|
||||
\lambda \vert\vert \boldsymbol{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert.
|
||||
$$
|
||||
|
||||
<p>It means that, when scaling the design matrix and the outputs/targets,
|
||||
by subtracting the mean values, we have an optimization problem which
|
||||
is not penalized by the intercept. The MSE value can then be smaller
|
||||
since it focuses only on the remaining quantities. If we however bring
|
||||
back the intercept, we will get a MSE which then contains the
|
||||
intercept.
|
||||
</p>
|
||||
|
||||
<p>Armed with this wisdom, we attempt first to simply set the intercept equal to <b>False</b> in our implementation of Ridge regression for our well-known vanilla data set.</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree))
|
||||
<span style="color: #408080; font-style: italic">#We include explicitely the intercept column</span>
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Maxpolydegree):
|
||||
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">6</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">2</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #408080; font-style: italic"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb,fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
ytildeOwnRidge <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ytildeRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSERidgePredict[i])
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'r'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.
|
||||
What happens if we do not include the intercept in our fit?
|
||||
Let us see how we can change this code by zero centering.
|
||||
</p>
|
||||
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="cell border-box-sizing code_cell rendered">
|
||||
<div class="input">
|
||||
<div class="inner_cell">
|
||||
<div class="input_area">
|
||||
<div class="highlight" style="background: #f8f8f8">
|
||||
<pre style="line-height: 125%;"><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
<span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
<span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)</span>
|
||||
<span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">6</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">2</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #008000">print</span>(intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSEOwnRidgePredict[i])
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"MSE values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(MSERidgePredict[i])
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
<div class="output_wrapper">
|
||||
<div class="output">
|
||||
<div class="output_area">
|
||||
<div class="output_subarea output_stream output_stdout output_text">
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<p>We see here, when compared to the code which includes explicitely the
|
||||
intercept column, that our MSE value is actually smaller. This is
|
||||
because the regularization term does not include the intercept value
|
||||
\( \beta_0 \) in the fitting. This applies to Lasso regularization as
|
||||
well. It means that our optimization is now done only with the
|
||||
centered matrix and/or vector that enter the fitting procedure.
|
||||
</p>
|
||||
|
||||
<!-- !split -->
|
||||
<h2 id="linking-the-regression-analysis-with-a-statistical-interpretation">Linking the regression analysis with a statistical interpretation </h2>
|
||||
|
||||
Binary file not shown.
+1851
-622
File diff suppressed because one or more lines are too long
File diff suppressed because it is too large
Load Diff
@@ -11,7 +11,8 @@ DATE: September 4-8, 2023
|
||||
* Recommended Reading: Hastie et al chapter 3, see URL:"https://link.springer.com/book/10.1007/978-0-387-84858-7"
|
||||
* Presentation and discussion of first project
|
||||
* Material for the lecture on Thursday September 7
|
||||
* Linear Regression and links with Statistics, Resampling methods
|
||||
* Technicalities related to scaling and other issues with data handling
|
||||
* Linear Regression and links with Statistics
|
||||
* Recommended Reading: Goodfellow et al chapter 3 on probability theory, see URL:""
|
||||
* See also Murphy, sections 2.4 (Gaussian distributions) and 3.2 (Bayesian Statistics, basis)
|
||||
|
||||
@@ -923,6 +924,499 @@ plt.show()
|
||||
===== Material for lecture Thursday September 7 =====
|
||||
|
||||
|
||||
===== Important technicalities: More on Rescaling data =====
|
||||
|
||||
|
||||
When you are comparing your own code with for example _Scikit-Learn_'s
|
||||
library, there are some technicalities to keep in mind. The examples
|
||||
here demonstrate some of these aspects with potential pitfalls.
|
||||
|
||||
The discussion here focuses on the role of the intercept, how we can
|
||||
set up the design matrix, what scaling we should use and other topics
|
||||
which tend confuse us.
|
||||
|
||||
The intercept can be interpreted as the expected value of our
|
||||
target/output variables when all other predictors are set to zero.
|
||||
Thus, if we cannot assume that the expected outputs/targets are zero
|
||||
when all predictors are zero (the columns in the design matrix), it
|
||||
may be a bad idea to implement a model which penalizes the intercept.
|
||||
Furthermore, in for example Ridge and Lasso regression, the default solutions
|
||||
from the library _Scikit-Learn_ (when not shrinking $\beta_0$) for the unknown parameters
|
||||
$\bm{\beta}$, are derived under the assumption that both $\bm{y}$ and
|
||||
$\bm{X}$ are zero centered, that is we subtract the mean values.
|
||||
|
||||
|
||||
If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix $\bm{X}$ by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation. Most machine learning libraries do this as a default. This means that if you compare your code with the results from a given library,
|
||||
the results may differ.
|
||||
|
||||
The
|
||||
"Standardscaler":"https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html"
|
||||
function in _Scikit-Learn_ does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
|
||||
If you need to scale the data, not doing so will give an *unfair*
|
||||
penalization of the parameters since their magnitude depends on the
|
||||
scale of their corresponding predictor.
|
||||
|
||||
Suppose as an example that you
|
||||
you have an input variable given by the heights of different persons.
|
||||
Human height might be measured in inches or meters or
|
||||
kilometers. If measured in kilometers, a standard linear regression
|
||||
model with this predictor would probably give a much bigger
|
||||
coefficient term, than if measured in millimeters.
|
||||
This can clearly lead to problems in evaluating the cost/loss functions.
|
||||
|
||||
|
||||
|
||||
Keep in mind that when you transform your data set before training a model, the same transformation needs to be done
|
||||
on your eventual new data set before making a prediction. If we translate this into a Python code, it would could be implemented as
|
||||
|
||||
!bc pycod
|
||||
"""
|
||||
#Model training, we compute the mean value of y and X
|
||||
y_train_mean = np.mean(y_train)
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
X_train = X_train - X_train_mean
|
||||
y_train = y_train - y_train_mean
|
||||
|
||||
# The we fit our model with the training data
|
||||
trained_model = some_model.fit(X_train,y_train)
|
||||
|
||||
|
||||
#Model prediction, we need also to transform our data set used for the prediction.
|
||||
X_test = X_test - X_train_mean #Use mean from training data
|
||||
y_pred = trained_model(X_test)
|
||||
y_pred = y_pred + y_train_mean
|
||||
"""
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
Let us try to understand what this may imply mathematically when we
|
||||
subtract the mean values, also known as *zero centering*. For
|
||||
simplicity, we will focus on ordinary regression, as done in the above example.
|
||||
|
||||
The cost/loss function for regression is
|
||||
!bt
|
||||
\[
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \frac{1}{n}\sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2,.
|
||||
\]
|
||||
!et
|
||||
|
||||
Recall also that we use the squared value. This expression can lead to an
|
||||
increased penalty for higher differences between predicted and
|
||||
output/target values.
|
||||
|
||||
What we have done is to single out the $\beta_0$ term in the
|
||||
definition of the mean squared error (MSE). The design matrix $X$
|
||||
does in this case not contain any intercept column. When we take the
|
||||
derivative with respect to $\beta_0$, we want the derivative to obey
|
||||
|
||||
!bt
|
||||
\[
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
\]
|
||||
!et
|
||||
|
||||
for all $j$. For $\beta_0$ we have
|
||||
|
||||
!bt
|
||||
\[
|
||||
\frac{\partial C}{\partial \beta_0} = -\frac{2}{n}\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right).
|
||||
\]
|
||||
!et
|
||||
Multiplying away the constant $2/n$, we obtain
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
\]
|
||||
!et
|
||||
|
||||
Let us specialize first to the case where we have only two parameters $\beta_0$ and $\beta_1$.
|
||||
Our result for $\beta_0$ simplifies then to
|
||||
!bt
|
||||
\[
|
||||
n\beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} X_{i1} \beta_1.
|
||||
\]
|
||||
!et
|
||||
We obtain then
|
||||
!bt
|
||||
\[
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \beta_1\frac{1}{n}\sum_{i=0}^{n-1} X_{i1}.
|
||||
\]
|
||||
!et
|
||||
If we define
|
||||
!bt
|
||||
\[
|
||||
\mu_{\bm{x}_1}=\frac{1}{n}\sum_{i=0}^{n-1} X_{i1},
|
||||
\]
|
||||
!et
|
||||
and the mean value of the outputs as
|
||||
!bt
|
||||
\[
|
||||
\mu_y=\frac{1}{n}\sum_{i=0}^{n-1}y_i,
|
||||
\]
|
||||
!et
|
||||
we have
|
||||
!bt
|
||||
\[
|
||||
\beta_0 = \mu_y - \beta_1\mu_{\bm{x}_1}.
|
||||
\]
|
||||
!et
|
||||
In the general case with more parameters than $\beta_0$ and $\beta_1$, we have
|
||||
!bt
|
||||
\[
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \frac{1}{n}\sum_{i=0}^{n-1}\sum_{j=1}^{p-1} X_{ij}\beta_j.
|
||||
\]
|
||||
!et
|
||||
|
||||
We can rewrite the latter equation as
|
||||
!bt
|
||||
\[
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1}y_i - \sum_{j=1}^{p-1} \mu_{\bm{x}_j}\beta_j,
|
||||
\]
|
||||
!et
|
||||
where we have defined
|
||||
!bt
|
||||
\[
|
||||
\mu_{\bm{x}_j}=\frac{1}{n}\sum_{i=0}^{n-1} X_{ij},
|
||||
\]
|
||||
!et
|
||||
the mean value for all elements of the column vector $\bm{x}_j$.
|
||||
|
||||
|
||||
|
||||
Replacing $y_i$ with $y_i - y_i - \overline{\bm{y}}$ and centering also our design matrix results in a cost function (in vector-matrix disguise)
|
||||
!bt
|
||||
\[
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}).
|
||||
\]
|
||||
!et
|
||||
|
||||
|
||||
|
||||
If we minimize with respect to $\bm{\beta}$ we have then
|
||||
|
||||
!bt
|
||||
\[
|
||||
\hat{\bm{\beta}} = (\tilde{X}^T\tilde{X})^{-1}\tilde{X}^T\boldsymbol{\tilde{y}},
|
||||
\]
|
||||
!et
|
||||
|
||||
where $\boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\bm{y}}$
|
||||
and $\tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=0}^{n-1}X_{kj}$.
|
||||
|
||||
For Ridge regression we need to add $\lambda \boldsymbol{\beta}^T\boldsymbol{\beta}$ to the cost function and get then
|
||||
!bt
|
||||
\[
|
||||
\hat{\bm{\beta}} = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}.
|
||||
\]
|
||||
!et
|
||||
|
||||
What does this mean? And why do we insist on all this? Let us look at some examples.
|
||||
|
||||
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*). Here our scaling of the data is done by subtracting the mean values only.
|
||||
Note also that we do not split the data into training and test.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import matplotlib.pyplot as plt
|
||||
|
||||
from sklearn.linear_model import LinearRegression
|
||||
|
||||
|
||||
np.random.seed(2021)
|
||||
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
|
||||
def fit_beta(X, y):
|
||||
return np.linalg.pinv(X.T @ X) @ X.T @ y
|
||||
|
||||
|
||||
true_beta = [2, 0.5, 3.7]
|
||||
|
||||
x = np.linspace(0, 1, 11)
|
||||
y = np.sum(
|
||||
np.asarray([x ** p * b for p, b in enumerate(true_beta)]), axis=0
|
||||
) + 0.1 * np.random.normal(size=len(x))
|
||||
|
||||
degree = 3
|
||||
X = np.zeros((len(x), degree))
|
||||
|
||||
# Include the intercept in the design matrix
|
||||
for p in range(degree):
|
||||
X[:, p] = x ** p
|
||||
|
||||
beta = fit_beta(X, y)
|
||||
|
||||
# Intercept is included in the design matrix
|
||||
skl = LinearRegression(fit_intercept=False).fit(X, y)
|
||||
|
||||
print(f"True beta: {true_beta}")
|
||||
print(f"Fitted beta: {beta}")
|
||||
print(f"Sklearn fitted beta: {skl.coef_}")
|
||||
ypredictOwn = X @ beta
|
||||
ypredictSKL = skl.predict(X)
|
||||
print(f"MSE with intercept column")
|
||||
print(MSE(y,ypredictOwn))
|
||||
print(f"MSE with intercept column from SKL")
|
||||
print(MSE(y,ypredictSKL))
|
||||
|
||||
|
||||
plt.figure()
|
||||
plt.scatter(x, y, label="Data")
|
||||
plt.plot(x, X @ beta, label="Fit")
|
||||
plt.plot(x, skl.predict(X), label="Sklearn (fit_intercept=False)")
|
||||
|
||||
|
||||
# Do not include the intercept in the design matrix
|
||||
X = np.zeros((len(x), degree - 1))
|
||||
|
||||
for p in range(degree - 1):
|
||||
X[:, p] = x ** (p + 1)
|
||||
|
||||
# Intercept is not included in the design matrix
|
||||
skl = LinearRegression(fit_intercept=True).fit(X, y)
|
||||
|
||||
# Use centered values for X and y when computing coefficients
|
||||
y_offset = np.average(y, axis=0)
|
||||
X_offset = np.average(X, axis=0)
|
||||
|
||||
beta = fit_beta(X - X_offset, y - y_offset)
|
||||
intercept = np.mean(y_offset - X_offset @ beta)
|
||||
|
||||
print(f"Manual intercept: {intercept}")
|
||||
print(f"Fitted beta (wiothout intercept): {beta}")
|
||||
print(f"Sklearn intercept: {skl.intercept_}")
|
||||
print(f"Sklearn fitted beta (without intercept): {skl.coef_}")
|
||||
ypredictOwn = X @ beta
|
||||
ypredictSKL = skl.predict(X)
|
||||
print(f"MSE with Manual intercept")
|
||||
print(MSE(y,ypredictOwn+intercept))
|
||||
print(f"MSE with Sklearn intercept")
|
||||
print(MSE(y,ypredictSKL))
|
||||
|
||||
plt.plot(x, X @ beta + intercept, "--", label="Fit (manual intercept)")
|
||||
plt.plot(x, skl.predict(X), "--", label="Sklearn (fit_intercept=True)")
|
||||
plt.grid()
|
||||
plt.legend()
|
||||
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
The intercept is the value of our output/target variable
|
||||
when all our features are zero and our function crosses the $y$-axis (for a one-dimensional case).
|
||||
|
||||
Printing the MSE, we see first that both methods give the same MSE, as
|
||||
they should. However, when we move to for example Ridge regression,
|
||||
the way we treat the intercept may give a larger or smaller MSE,
|
||||
meaning that the MSE can be penalized by the value of the
|
||||
intercept. Not including the intercept in the fit, means that the
|
||||
regularization term does not include $\beta_0$. For different values
|
||||
of $\lambda$, this may lead to different MSE values.
|
||||
|
||||
To remind the reader, the regularization term, with the intercept in Ridge regression, is given by
|
||||
!bt
|
||||
\[
|
||||
\lambda \vert\vert \bm{\beta} \vert\vert_2^2 = \lambda \sum_{j=0}^{p-1}\beta_j^2,
|
||||
\]
|
||||
!et
|
||||
but when we take out the intercept, this equation becomes
|
||||
!bt
|
||||
\[
|
||||
\lambda \vert\vert \bm{\beta} \vert\vert_2^2 = \lambda \sum_{j=1}^{p-1}\beta_j^2.
|
||||
\]
|
||||
!et
|
||||
|
||||
For Lasso regression we have
|
||||
!bt
|
||||
\[
|
||||
\lambda \vert\vert \bm{\beta} \vert\vert_1 = \lambda \sum_{j=1}^{p-1}\vert\beta_j\vert.
|
||||
\]
|
||||
!et
|
||||
|
||||
It means that, when scaling the design matrix and the outputs/targets,
|
||||
by subtracting the mean values, we have an optimization problem which
|
||||
is not penalized by the intercept. The MSE value can then be smaller
|
||||
since it focuses only on the remaining quantities. If we however bring
|
||||
back the intercept, we will get a MSE which then contains the
|
||||
intercept.
|
||||
|
||||
|
||||
Armed with this wisdom, we attempt first to simply set the intercept equal to _False_ in our implementation of Ridge regression for our well-known vanilla data set.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn import linear_model
|
||||
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(3155)
|
||||
|
||||
n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 20
|
||||
X = np.zeros((n,Maxpolydegree))
|
||||
#We include explicitely the intercept column
|
||||
for degree in range(Maxpolydegree):
|
||||
X[:,degree] = x**degree
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
p = Maxpolydegree
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 6
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
lambdas = np.logspace(-4, 2, nlambdas)
|
||||
for i in range(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
|
||||
# Note: we include the intercept column and no scaling
|
||||
RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
# and then make the prediction
|
||||
ytildeOwnRidge = X_train @ OwnRidgeBeta
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta
|
||||
ytildeRidge = RegRidge.predict(X_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
print("Beta values for own Ridge implementation")
|
||||
print(OwnRidgeBeta)
|
||||
print("Beta values for Scikit-Learn Ridge implementation")
|
||||
print(RegRidge.coef_)
|
||||
print("MSE values for own Ridge implementation")
|
||||
print(MSEOwnRidgePredict[i])
|
||||
print("MSE values for Scikit-Learn Ridge implementation")
|
||||
print(MSERidgePredict[i])
|
||||
|
||||
# Now plot the results
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE own Ridge Test')
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')
|
||||
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
The results here agree when we force _Scikit-Learn_'s Ridge function to include the first column in our design matrix.
|
||||
We see that the results agree very well. Here we have thus explicitely included the intercept column in the design matrix.
|
||||
What happens if we do not include the intercept in our fit?
|
||||
Let us see how we can change this code by zero centering.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn import linear_model
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(315)
|
||||
|
||||
n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 20
|
||||
X = np.zeros((n,Maxpolydegree-1))
|
||||
|
||||
for degree in range(1,Maxpolydegree): #No intercept column
|
||||
X[:,degree-1] = x**(degree)
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
#Center by removing mean from each feature
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
#The model intercept (called y_scaler) is given by the mean of the target variable (IF X is centered)
|
||||
#Remove the intercept from the training data.
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
p = Maxpolydegree-1
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 6
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-4, 2, nlambdas)
|
||||
for i in range(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
|
||||
#Add intercept to prediction
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
print("Beta values for own Ridge implementation")
|
||||
print(OwnRidgeBeta) #Intercept is given by mean of target variable
|
||||
print("Beta values for Scikit-Learn Ridge implementation")
|
||||
print(RegRidge.coef_)
|
||||
print('Intercept from own implementation:')
|
||||
print(intercept_)
|
||||
print('Intercept from Scikit-Learn Ridge implementation')
|
||||
print(RegRidge.intercept_)
|
||||
print("MSE values for own Ridge implementation")
|
||||
print(MSEOwnRidgePredict[i])
|
||||
print("MSE values for Scikit-Learn Ridge implementation")
|
||||
print(MSERidgePredict[i])
|
||||
|
||||
|
||||
# Now plot the results
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
!ec
|
||||
We see here, when compared to the code which includes explicitely the
|
||||
intercept column, that our MSE value is actually smaller. This is
|
||||
because the regularization term does not include the intercept value
|
||||
$\beta_0$ in the fitting. This applies to Lasso regularization as
|
||||
well. It means that our optimization is now done only with the
|
||||
centered matrix and/or vector that enter the fitting procedure.
|
||||
|
||||
|
||||
!split
|
||||
===== Linking the regression analysis with a statistical interpretation =====
|
||||
|
||||
@@ -1567,3 +2061,5 @@ which is our Lasso cost function!
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user