more update
This commit is contained in:
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -452,7 +457,7 @@ MathJax.Hub.Config({
|
||||
<li><a href="._week38-bs008.html">9</a></li>
|
||||
<li><a href="._week38-bs009.html">10</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs001.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -432,7 +437,7 @@ MathJax.Hub.Config({
|
||||
<li><a href="._week38-bs009.html">10</a></li>
|
||||
<li><a href="._week38-bs010.html">11</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs002.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -429,7 +434,7 @@ MathJax.Hub.Config({
|
||||
<li><a href="._week38-bs010.html">11</a></li>
|
||||
<li><a href="._week38-bs011.html">12</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs003.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -483,7 +488,7 @@ $$
|
||||
<li><a href="._week38-bs011.html">12</a></li>
|
||||
<li><a href="._week38-bs012.html">13</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs004.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -447,7 +452,7 @@ cross-validation (LOOCV).
|
||||
<li><a href="._week38-bs012.html">13</a></li>
|
||||
<li><a href="._week38-bs013.html">14</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs005.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -459,7 +464,7 @@ $$
|
||||
<li><a href="._week38-bs013.html">14</a></li>
|
||||
<li><a href="._week38-bs014.html">15</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs006.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -450,7 +455,7 @@ For the various values of \( k \)
|
||||
<li><a href="._week38-bs014.html">15</a></li>
|
||||
<li><a href="._week38-bs015.html">16</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs007.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -529,7 +534,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs015.html">16</a></li>
|
||||
<li><a href="._week38-bs016.html">17</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs008.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -456,7 +461,7 @@ Furthermore, in for example Ridge and Lasso regression, the solutions
|
||||
<li><a href="._week38-bs016.html">17</a></li>
|
||||
<li><a href="._week38-bs017.html">18</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs009.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -415,14 +420,16 @@ MathJax.Hub.Config({
|
||||
If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation.
|
||||
standard deviation. Most machine learning libraries do this as a deafult. This means that if you compare your code with the results from a given library,
|
||||
the results may differ. Tracing back the differences may often lead to an increased confusion.
|
||||
|
||||
<p>
|
||||
The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_self">Standadscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them.
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
|
||||
<p>
|
||||
If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
@@ -463,7 +470,7 @@ This can clearly lead to problems in evaluating the cost/loss functions.
|
||||
<li><a href="._week38-bs017.html">18</a></li>
|
||||
<li><a href="._week38-bs018.html">19</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs010.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -459,7 +464,7 @@ y_pred <span style="color: #666666">=</span> y_pred <span style="color: #666666"
|
||||
<li><a href="._week38-bs018.html">19</a></li>
|
||||
<li><a href="._week38-bs019.html">20</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs011.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -412,7 +417,7 @@ MathJax.Hub.Config({
|
||||
<h2 id="linear-regression-code-intercept-handling-first" class="anchor">Linear Regression code, Intercept handling first </h2>
|
||||
|
||||
<p>
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>)
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
|
||||
<p>
|
||||
|
||||
@@ -514,7 +519,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs019.html">20</a></li>
|
||||
<li><a href="._week38-bs020.html">21</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs012.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,81 +414,72 @@ MathJax.Hub.Config({
|
||||
<a name="part0012"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="what-does-centering-mean-mathematically" class="anchor">What does centering mean mathematically? </h2>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
<h2 id="what-does-centering-subtracting-the-mean-values-mean-mathematically" class="anchor">What does centering (subtracting the mean values) mean mathematically? </h2>
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
Let us try to understand what this may imply mathematically when we subtract the mean values, also known as <em>zero centering</em>. To catch many birds with just one stone, we will focus on Ridge regression.
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2 + \lambda \sum_{j=1}^{p-1} \beta_i^2.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
Note that the intercept term $\beta_0$is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
for all \( j \), so let us start with \( \beta_0 \). This means that we have
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
\frac{\partial C}{\partial \beta_0} = -2\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right),
|
||||
$$
|
||||
|
||||
<p>
|
||||
which gives
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
If we assume that every column of \( \boldsymbol{X} \) is centered, which we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> X <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(X,axis<span style="color: #666666">=0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
the sum \( \sum_{i=0}^{n-1} X_{ij} \)
|
||||
can be rewritten as
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
\sum_{i=0}^{n-1} \left(X_{ij} - \frac{1}{n}\sum_{i=0}^{n-1} X_{ij}) = \sum_{i=0}^{n-1} X_{ij} - \sum_{i=0}^{n-1} \frac{1}{n} \sum_{i=0}^{n-1}X_{ij},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
\sum_{i=0}^{n-1} X_{ij} - n \frac{1}{n} \sum_{i=0}^{n-1}X_{ij} = 0.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
n\beta_0 = \sum_{i=0}^{n-1} y_i - \sum_{j=1}^{p-1}\beta_j \sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
|
||||
or
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1} y_i = \overline{\boldsymbol{y}},
|
||||
$$
|
||||
|
||||
the average value of \( \boldsymbol{y] \).
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - \overline{\boldsymbol{y}} \) in the cost function will give us (in vector-matrix disguise)
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
@@ -493,201 +489,9 @@ which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">R2</span>(y_data, y_model):
|
||||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span> <span style="color: #666666">-</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> y_model) <span style="color: #666666">**</span> <span style="color: #666666">2</span>) <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(y_data)) <span style="color: #666666">**</span> <span style="color: #666666">2</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree))
|
||||
X[:,<span style="color: #666666">0</span>] <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> polydegree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>, Maxpolydegree):
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(polydegree):
|
||||
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># matrix inversion to find beta</span>
|
||||
OLSbeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #008000">print</span>(OLSbeta)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
ytildeOLS <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OLSbeta
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Training MSE for OLS"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y_train,ytildeOLS))
|
||||
ypredictOLS <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OLSbeta
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test MSE OLS"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y_test,ypredictOLS))
|
||||
|
||||
p <span style="color: #666666">=</span> <span style="color: #008000">len</span>(OLSbeta)
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSEOwnRidgeTrain <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgeTrain <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">4</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #408080; font-style: italic"># include lasso using Scikit-Learn</span>
|
||||
<span style="color: #408080; font-style: italic"># Note: we include the intercept</span>
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb,fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
ytildeOwnRidge <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ytildeRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSEOwnRidgeTrain[i] <span style="color: #666666">=</span> MSE(y_train,ytildeOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
MSERidgeTrain[i] <span style="color: #666666">=</span> MSE(y_train,ytildeRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgeTrain, <span style="color: #BA2121">'b'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge train'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'r'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgeTrain, <span style="color: #BA2121">'y'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge train'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">R2</span>(y_data, y_model):
|
||||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span> <span style="color: #666666">-</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> y_model) <span style="color: #666666">**</span> <span style="color: #666666">2</span>) <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(y_data)) <span style="color: #666666">**</span> <span style="color: #666666">2</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">5</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean <span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train) <span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler <span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">1</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> intercept_ <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #408080; font-style: italic">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #008000">print</span>(intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -714,7 +518,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs020.html">21</a></li>
|
||||
<li><a href="._week38-bs021.html">22</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs013.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,59 +414,87 @@ MathJax.Hub.Config({
|
||||
<a name="part0013"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="more-complicated-example-the-ising-model" class="anchor">More complicated Example: The Ising model </h2>
|
||||
<h2 id="code-examples" class="anchor">Code Examples </h2>
|
||||
|
||||
<p>
|
||||
The one-dimensional Ising model with nearest neighbor interaction, no
|
||||
external field and a constant coupling constant \( J \) is given by
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
H = -J \sum_{k}^L s_k s_{k + 1},
|
||||
\tag{1}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins
|
||||
in the system is determined by \( L \). For the one-dimensional system
|
||||
there is no phase transition.
|
||||
|
||||
<p>
|
||||
We will look at a system of \( L = 40 \) spins with a coupling constant of
|
||||
\( J = 1 \). To get enough training data we will generate 10000 states
|
||||
with their respective energies.
|
||||
Armed with this wisdom, we attempt first simply set the intercept eqault to <b>False</b> in our implementation of Ridge regression for a vanilla data set.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">mpl_toolkits.axes_grid1</span> <span style="color: #008000; font-weight: bold">import</span> make_axes_locatable
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">seaborn</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">sns</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">scipy.linalg</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">scl</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">tqdm</span>
|
||||
sns<span style="color: #666666">.</span>set(color_codes<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)
|
||||
cmap_args<span style="color: #666666">=</span><span style="color: #008000">dict</span>(vmin<span style="color: #666666">=-1.</span>, vmax<span style="color: #666666">=1.</span>, cmap<span style="color: #666666">=</span><span style="color: #BA2121">'seismic'</span>)
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
|
||||
L <span style="color: #666666">=</span> <span style="color: #666666">40</span>
|
||||
n <span style="color: #666666">=</span> <span style="color: #008000">int</span>(<span style="color: #666666">1e4</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
spins <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>choice([<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>], size<span style="color: #666666">=</span>(n, L))
|
||||
J <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
energies <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(n)
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">3155</span>)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
energies[i] <span style="color: #666666">=</span> <span style="color: #666666">-</span> J <span style="color: #666666">*</span> np<span style="color: #666666">.</span>dot(spins[i], np<span style="color: #666666">.</span>roll(spins[i], <span style="color: #666666">1</span>))
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree))
|
||||
We include explicitely the intercpt column
|
||||
X[:,<span style="color: #666666">0</span>] <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Maxpolydegree):
|
||||
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">4</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #408080; font-style: italic"># include lasso using Scikit-Learn</span>
|
||||
<span style="color: #408080; font-style: italic"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb,fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
ytildeOwnRidge <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta
|
||||
ytildeRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'r'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
Here we use ordinary least squares
|
||||
regression to predict the energy for the nearest neighbor
|
||||
one-dimensional Ising model on a ring, i.e., the endpoints wrap
|
||||
around. We will use linear regression to fit a value for
|
||||
the coupling constant to achieve this.
|
||||
The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
The problem however is that can easily lead to a larger mean-squared error!
|
||||
|
||||
<p>
|
||||
Let us see how we can change this code by zero centering.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -489,7 +522,7 @@ the coupling constant to achieve this.
|
||||
<li><a href="._week38-bs021.html">22</a></li>
|
||||
<li><a href="._week38-bs022.html">23</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs014.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,52 +414,95 @@ MathJax.Hub.Config({
|
||||
<a name="part0014"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="reformulating-the-problem-to-suit-regression" class="anchor">Reformulating the problem to suit regression </h2>
|
||||
|
||||
<p>
|
||||
A more general form for the one-dimensional Ising model is
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
|
||||
\tag{2}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
Here we allow for interactions beyond the nearest neighbors and a state dependent
|
||||
coupling constant. This latter expression can be formulated as
|
||||
a matrix-product
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{H} = \boldsymbol{X} J,
|
||||
\tag{3}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the
|
||||
elements \( -J_{jk} \). This form of writing the energy fits perfectly
|
||||
with the form utilized in linear regression, that is
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon},
|
||||
\tag{4}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
We split the data in training and test data as discussed in the previous example
|
||||
|
||||
<h2 id="taking-out-the-mean" class="anchor">Taking out the mean </h2>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n, L <span style="color: #666666">**</span> <span style="color: #666666">2</span>))
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
X[i] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>outer(spins[i], spins[i])<span style="color: #666666">.</span>ravel()
|
||||
y <span style="color: #666666">=</span> energies
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
|
||||
n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
<span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
<span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
<span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler
|
||||
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">1</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> intercept_
|
||||
<span style="color: #408080; font-style: italic">#EQUIVALENT PREDICTION:</span>
|
||||
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictOwnRidge)
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictRidge)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from own implementation:'</span>)
|
||||
<span style="color: #008000">print</span>(intercept_)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
@@ -482,7 +530,7 @@ X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_tes
|
||||
<li><a href="._week38-bs022.html">23</a></li>
|
||||
<li><a href="._week38-bs023.html">24</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs015.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,51 +414,60 @@ MathJax.Hub.Config({
|
||||
<a name="part0015"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="linear-regression" class="anchor">Linear regression </h2>
|
||||
<h2 id="more-complicated-example-the-ising-model" class="anchor">More complicated Example: The Ising model </h2>
|
||||
|
||||
<p>
|
||||
In the ordinary least squares method we choose the cost function
|
||||
The one-dimensional Ising model with nearest neighbor interaction, no
|
||||
external field and a constant coupling constant \( J \) is given by
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}.
|
||||
\tag{5}
|
||||
H = -J \sum_{k}^L s_k s_{k + 1},
|
||||
\tag{1}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above.
|
||||
This yields the expression for \( \boldsymbol{\beta} \) to be
|
||||
|
||||
$$
|
||||
\boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}},
|
||||
$$
|
||||
where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins
|
||||
in the system is determined by \( L \). For the one-dimensional system
|
||||
there is no phase transition.
|
||||
|
||||
<p>
|
||||
which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist
|
||||
an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an
|
||||
intercept, i.e., a constant term, we must make sure that the
|
||||
first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here
|
||||
We will look at a system of \( L = 40 \) spins with a coupling constant of
|
||||
\( J = 1 \). To get enough training data we will generate 10000 states
|
||||
with their respective energies.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X_train_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
(np<span style="color: #666666">.</span>ones(<span style="color: #008000">len</span>(X_train))[:, np<span style="color: #666666">.</span>newaxis], X_train),
|
||||
axis<span style="color: #666666">=1</span>
|
||||
)
|
||||
X_test_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
(np<span style="color: #666666">.</span>ones(<span style="color: #008000">len</span>(X_test))[:, np<span style="color: #666666">.</span>newaxis], X_test),
|
||||
axis<span style="color: #666666">=1</span>
|
||||
)
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">mpl_toolkits.axes_grid1</span> <span style="color: #008000; font-weight: bold">import</span> make_axes_locatable
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">seaborn</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">sns</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">scipy.linalg</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">scl</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">tqdm</span>
|
||||
sns<span style="color: #666666">.</span>set(color_codes<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)
|
||||
cmap_args<span style="color: #666666">=</span><span style="color: #008000">dict</span>(vmin<span style="color: #666666">=-1.</span>, vmax<span style="color: #666666">=1.</span>, cmap<span style="color: #666666">=</span><span style="color: #BA2121">'seismic'</span>)
|
||||
|
||||
L <span style="color: #666666">=</span> <span style="color: #666666">40</span>
|
||||
n <span style="color: #666666">=</span> <span style="color: #008000">int</span>(<span style="color: #666666">1e4</span>)
|
||||
|
||||
spins <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>choice([<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>], size<span style="color: #666666">=</span>(n, L))
|
||||
J <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
energies <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(n)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
energies[i] <span style="color: #666666">=</span> <span style="color: #666666">-</span> J <span style="color: #666666">*</span> np<span style="color: #666666">.</span>dot(spins[i], np<span style="color: #666666">.</span>roll(spins[i], <span style="color: #666666">1</span>))
|
||||
</pre></div>
|
||||
<p>
|
||||
Here we use ordinary least squares
|
||||
regression to predict the energy for the nearest neighbor
|
||||
one-dimensional Ising model on a ring, i.e., the endpoints wrap
|
||||
around. We will use linear regression to fit a value for
|
||||
the coupling constant to achieve this.
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">ols_inv</span>(x: np<span style="color: #666666">.</span>ndarray, y: np<span style="color: #666666">.</span>ndarray) <span style="color: #666666">-></span> np<span style="color: #666666">.</span>ndarray:
|
||||
<span style="color: #008000; font-weight: bold">return</span> scl<span style="color: #666666">.</span>inv(x<span style="color: #666666">.</span>T <span style="color: #666666">@</span> x) <span style="color: #666666">@</span> (x<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
||||
beta <span style="color: #666666">=</span> ols_inv(X_train_own, y_train)
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -480,7 +494,7 @@ beta <span style="color: #666666">=</span> ols_inv(X_train_own, y_train)
|
||||
<li><a href="._week38-bs023.html">24</a></li>
|
||||
<li><a href="._week38-bs024.html">25</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs016.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,90 +414,53 @@ MathJax.Hub.Config({
|
||||
<a name="part0016"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="singular-value-decomposition" class="anchor">Singular Value decomposition </h2>
|
||||
<h2 id="reformulating-the-problem-to-suit-regression" class="anchor">Reformulating the problem to suit regression </h2>
|
||||
|
||||
<p>
|
||||
Doing the inversion directly turns out to be a bad idea since the matrix
|
||||
\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the <b>singular
|
||||
value decomposition</b>. Using the definition of the Moore-Penrose
|
||||
pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as
|
||||
A more general form for the one-dimensional Ising model is
|
||||
|
||||
$$
|
||||
\boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y},
|
||||
$$
|
||||
|
||||
<p>
|
||||
where the pseudoinverse of \( \boldsymbol{X} \) is given by
|
||||
|
||||
$$
|
||||
\boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \),
|
||||
where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below).
|
||||
where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for
|
||||
\( \omega \) to
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}.
|
||||
\tag{6}
|
||||
H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
|
||||
\tag{2}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
Note that solving this equation by actually doing the pseudoinverse
|
||||
(which is what we will do) is not a good idea as this operation scales
|
||||
as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a
|
||||
general matrix. Instead, doing \( QR \)-factorization and solving the
|
||||
linear system as an equation would reduce this down to
|
||||
\( \mathcal{O}(n^2) \) operations.
|
||||
Here we allow for interactions beyond the nearest neighbors and a state dependent
|
||||
coupling constant. This latter expression can be formulated as
|
||||
a matrix-product
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{H} = \boldsymbol{X} J,
|
||||
\tag{3}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
where \( X_{jk} = s_j s_k \) and \( J \) is a matrix which consists of the
|
||||
elements \( -J_{jk} \). This form of writing the energy fits perfectly
|
||||
with the form utilized in linear regression, that is
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon},
|
||||
\tag{4}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
We split the data in training and test data as discussed in the previous example
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">ols_svd</span>(x: np<span style="color: #666666">.</span>ndarray, y: np<span style="color: #666666">.</span>ndarray) <span style="color: #666666">-></span> np<span style="color: #666666">.</span>ndarray:
|
||||
u, s, v <span style="color: #666666">=</span> scl<span style="color: #666666">.</span>svd(x)
|
||||
<span style="color: #008000; font-weight: bold">return</span> v<span style="color: #666666">.</span>T <span style="color: #666666">@</span> scl<span style="color: #666666">.</span>pinv(scl<span style="color: #666666">.</span>diagsvd(s, u<span style="color: #666666">.</span>shape[<span style="color: #666666">0</span>], v<span style="color: #666666">.</span>shape[<span style="color: #666666">0</span>])) <span style="color: #666666">@</span> u<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n, L <span style="color: #666666">**</span> <span style="color: #666666">2</span>))
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
X[i] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>outer(spins[i], spins[i])<span style="color: #666666">.</span>ravel()
|
||||
y <span style="color: #666666">=</span> energies
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>beta <span style="color: #666666">=</span> ols_svd(X_train_own,y_train)
|
||||
</pre></div>
|
||||
<p>
|
||||
When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>J <span style="color: #666666">=</span> beta[<span style="color: #666666">1</span>:]<span style="color: #666666">.</span>reshape(L, L)
|
||||
</pre></div>
|
||||
<p>
|
||||
A way of looking at the coefficients in \( J \) is to plot the matrices as images.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"OLS"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xticks(fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>yticks(fontsize<span style="color: #666666">=18</span>)
|
||||
cb <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>colorbar(im)
|
||||
cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>set_yticklabels(cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>get_yticklabels(), fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
It is interesting to note that OLS
|
||||
considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as
|
||||
valid matrix elements for \( J \).
|
||||
In our discussion below on hyperparameters and Ridge and Lasso regression we will see that
|
||||
this problem can be removed, partly and only with Lasso regression.
|
||||
|
||||
<p>
|
||||
In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD?
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -519,7 +487,7 @@ In this case our matrix inversion was actually possible. The obvious question no
|
||||
<li><a href="._week38-bs024.html">25</a></li>
|
||||
<li><a href="._week38-bs025.html">26</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs017.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,137 +414,51 @@ MathJax.Hub.Config({
|
||||
<a name="part0017"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="the-one-dimensional-ising-model" class="anchor">The one-dimensional Ising model </h2>
|
||||
<h2 id="linear-regression" class="anchor">Linear regression </h2>
|
||||
|
||||
<p>
|
||||
Let us bring back the Ising model again, but now with an additional
|
||||
focus on Ridge and Lasso regression as well. We repeat some of the
|
||||
basic parts of the Ising model and the setup of the training and test
|
||||
data. The one-dimensional Ising model with nearest neighbor
|
||||
interaction, no external field and a constant coupling constant \( J \) is
|
||||
given by
|
||||
In the ordinary least squares method we choose the cost function
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
H = -J \sum_{k}^L s_k s_{k + 1},
|
||||
\tag{7}
|
||||
C(\boldsymbol{X}, \boldsymbol{\beta})= \frac{1}{n}\left\{(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})\right\}.
|
||||
\tag{5}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition.
|
||||
<p>
|
||||
We then find the extremal point of \( C \) by taking the derivative with respect to \( \boldsymbol{\beta} \) as discussed above.
|
||||
This yields the expression for \( \boldsymbol{\beta} \) to be
|
||||
|
||||
$$
|
||||
\boldsymbol{\beta} = \frac{\boldsymbol{X}^T \boldsymbol{y}}{\boldsymbol{X}^T \boldsymbol{X}},
|
||||
$$
|
||||
|
||||
<p>
|
||||
We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies.
|
||||
which immediately imposes some requirements on \( \boldsymbol{X} \) as there must exist
|
||||
an inverse of \( \boldsymbol{X}^T \boldsymbol{X} \). If the expression we are modeling contains an
|
||||
intercept, i.e., a constant term, we must make sure that the
|
||||
first column of \( \boldsymbol{X} \) consists of \( 1 \). We do this here
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">mpl_toolkits.axes_grid1</span> <span style="color: #008000; font-weight: bold">import</span> make_axes_locatable
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">seaborn</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">sns</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">scipy.linalg</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">scl</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">skl</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">tqdm</span>
|
||||
sns<span style="color: #666666">.</span>set(color_codes<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)
|
||||
cmap_args<span style="color: #666666">=</span><span style="color: #008000">dict</span>(vmin<span style="color: #666666">=-1.</span>, vmax<span style="color: #666666">=1.</span>, cmap<span style="color: #666666">=</span><span style="color: #BA2121">'seismic'</span>)
|
||||
|
||||
L <span style="color: #666666">=</span> <span style="color: #666666">40</span>
|
||||
n <span style="color: #666666">=</span> <span style="color: #008000">int</span>(<span style="color: #666666">1e4</span>)
|
||||
|
||||
spins <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>choice([<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>], size<span style="color: #666666">=</span>(n, L))
|
||||
J <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
energies <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(n)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
energies[i] <span style="color: #666666">=</span> <span style="color: #666666">-</span> J <span style="color: #666666">*</span> np<span style="color: #666666">.</span>dot(spins[i], np<span style="color: #666666">.</span>roll(spins[i], <span style="color: #666666">1</span>))
|
||||
</pre></div>
|
||||
<p>
|
||||
A more general form for the one-dimensional Ising model is
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
|
||||
\tag{8}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
Here we allow for interactions beyond the nearest neighbors and a more
|
||||
adaptive coupling matrix. This latter expression can be formulated as
|
||||
a matrix-product on the form
|
||||
$$
|
||||
\begin{align}
|
||||
H = X J,
|
||||
\tag{9}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the
|
||||
elements \( -J_{jk} \). This form of writing the energy fits perfectly
|
||||
with the form utilized in linear regression, viz.
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}.
|
||||
\tag{10}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
We organize the data as we did above
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n, L <span style="color: #666666">**</span> <span style="color: #666666">2</span>))
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
X[i] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>outer(spins[i], spins[i])<span style="color: #666666">.</span>ravel()
|
||||
y <span style="color: #666666">=</span> energies
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.96</span>)
|
||||
|
||||
X_train_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X_train_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
(np<span style="color: #666666">.</span>ones(<span style="color: #008000">len</span>(X_train))[:, np<span style="color: #666666">.</span>newaxis], X_train),
|
||||
axis<span style="color: #666666">=1</span>
|
||||
)
|
||||
|
||||
X_test_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
(np<span style="color: #666666">.</span>ones(<span style="color: #008000">len</span>(X_test))[:, np<span style="color: #666666">.</span>newaxis], X_test),
|
||||
axis<span style="color: #666666">=1</span>
|
||||
)
|
||||
</pre></div>
|
||||
<p>
|
||||
We will do all fitting with <b>Scikit-Learn</b>,
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>clf <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>LinearRegression()<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">ols_inv</span>(x: np<span style="color: #666666">.</span>ndarray, y: np<span style="color: #666666">.</span>ndarray) <span style="color: #666666">-></span> np<span style="color: #666666">.</span>ndarray:
|
||||
<span style="color: #008000; font-weight: bold">return</span> scl<span style="color: #666666">.</span>inv(x<span style="color: #666666">.</span>T <span style="color: #666666">@</span> x) <span style="color: #666666">@</span> (x<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y)
|
||||
beta <span style="color: #666666">=</span> ols_inv(X_train_own, y_train)
|
||||
</pre></div>
|
||||
<p>
|
||||
When extracting the \( J \)-matrix we make sure to remove the intercept
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>J_sk <span style="color: #666666">=</span> clf<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
</pre></div>
|
||||
<p>
|
||||
And then we plot the results
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J_sk, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"LinearRegression from Scikit-learn"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xticks(fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>yticks(fontsize<span style="color: #666666">=18</span>)
|
||||
cb <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>colorbar(im)
|
||||
cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>set_yticklabels(cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>get_yticklabels(), fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
The results perfectly with our previous discussion where we used our own code.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -566,7 +485,7 @@ The results perfectly with our previous discussion where we used our own code.
|
||||
<li><a href="._week38-bs025.html">26</a></li>
|
||||
<li><a href="._week38-bs026.html">27</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs018.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,38 +414,90 @@ MathJax.Hub.Config({
|
||||
<a name="part0018"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="ridge-regression" class="anchor">Ridge regression </h2>
|
||||
<h2 id="singular-value-decomposition" class="anchor">Singular Value decomposition </h2>
|
||||
|
||||
<p>
|
||||
Having explored the ordinary least squares we move on to ridge
|
||||
regression. In ridge regression we include a <b>regularizer</b>. This
|
||||
involves a new cost function which leads to a new estimate for the
|
||||
weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The
|
||||
cost function is given by
|
||||
Doing the inversion directly turns out to be a bad idea since the matrix
|
||||
\( \boldsymbol{X}^T\boldsymbol{X} \) is singular. An alternative approach is to use the <b>singular
|
||||
value decomposition</b>. Using the definition of the Moore-Penrose
|
||||
pseudoinverse we can write the equation for \( \boldsymbol{\beta} \) as
|
||||
|
||||
$$
|
||||
\boldsymbol{\beta} = \boldsymbol{X}^{+}\boldsymbol{y},
|
||||
$$
|
||||
|
||||
<p>
|
||||
where the pseudoinverse of \( \boldsymbol{X} \) is given by
|
||||
|
||||
$$
|
||||
\boldsymbol{X}^{+} = \frac{\boldsymbol{X}^T}{\boldsymbol{X}^T\boldsymbol{X}}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Using singular value decomposition we can decompose the matrix \( \boldsymbol{X} = \boldsymbol{U}\boldsymbol{\Sigma} \boldsymbol{V}^T \),
|
||||
where \( \boldsymbol{U} \) and \( \boldsymbol{V} \) are orthogonal(unitary) matrices and \( \boldsymbol{\Sigma} \) contains the singular values (more details below).
|
||||
where \( X^{+} = V\Sigma^{+} U^T \). This reduces the equation for
|
||||
\( \omega \) to
|
||||
$$
|
||||
\begin{align}
|
||||
C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}.
|
||||
\tag{11}
|
||||
\boldsymbol{\beta} = \boldsymbol{V}\boldsymbol{\Sigma}^{+} \boldsymbol{U}^T \boldsymbol{y}.
|
||||
\tag{6}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
Note that solving this equation by actually doing the pseudoinverse
|
||||
(which is what we will do) is not a good idea as this operation scales
|
||||
as \( \mathcal{O}(n^3) \), where \( n \) is the number of elements in a
|
||||
general matrix. Instead, doing \( QR \)-factorization and solving the
|
||||
linear system as an equation would reduce this down to
|
||||
\( \mathcal{O}(n^2) \) operations.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>_lambda <span style="color: #666666">=</span> <span style="color: #666666">0.1</span>
|
||||
clf_ridge <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>Ridge(alpha<span style="color: #666666">=</span>_lambda)<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
J_ridge_sk <span style="color: #666666">=</span> clf_ridge<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J_ridge_sk, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"Ridge from Scikit-learn"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">ols_svd</span>(x: np<span style="color: #666666">.</span>ndarray, y: np<span style="color: #666666">.</span>ndarray) <span style="color: #666666">-></span> np<span style="color: #666666">.</span>ndarray:
|
||||
u, s, v <span style="color: #666666">=</span> scl<span style="color: #666666">.</span>svd(x)
|
||||
<span style="color: #008000; font-weight: bold">return</span> v<span style="color: #666666">.</span>T <span style="color: #666666">@</span> scl<span style="color: #666666">.</span>pinv(scl<span style="color: #666666">.</span>diagsvd(s, u<span style="color: #666666">.</span>shape[<span style="color: #666666">0</span>], v<span style="color: #666666">.</span>shape[<span style="color: #666666">0</span>])) <span style="color: #666666">@</span> u<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y
|
||||
</pre></div>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>beta <span style="color: #666666">=</span> ols_svd(X_train_own,y_train)
|
||||
</pre></div>
|
||||
<p>
|
||||
When extracting the \( J \)-matrix we need to make sure that we remove the intercept, as is done here
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>J <span style="color: #666666">=</span> beta[<span style="color: #666666">1</span>:]<span style="color: #666666">.</span>reshape(L, L)
|
||||
</pre></div>
|
||||
<p>
|
||||
A way of looking at the coefficients in \( J \) is to plot the matrices as images.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"OLS"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xticks(fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>yticks(fontsize<span style="color: #666666">=18</span>)
|
||||
cb <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>colorbar(im)
|
||||
cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>set_yticklabels(cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>get_yticklabels(), fontsize<span style="color: #666666">=18</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
It is interesting to note that OLS
|
||||
considers both \( J_{j, j + 1} = -0.5 \) and \( J_{j, j - 1} = -0.5 \) as
|
||||
valid matrix elements for \( J \).
|
||||
In our discussion below on hyperparameters and Ridge and Lasso regression we will see that
|
||||
this problem can be removed, partly and only with Lasso regression.
|
||||
|
||||
<p>
|
||||
In this case our matrix inversion was actually possible. The obvious question now is what is the mathematics behind the SVD?
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -467,7 +524,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs026.html">27</a></li>
|
||||
<li><a href="._week38-bs027.html">28</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs019.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,40 +414,136 @@ MathJax.Hub.Config({
|
||||
<a name="part0019"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="lasso-regression" class="anchor">LASSO regression </h2>
|
||||
<h2 id="the-one-dimensional-ising-model" class="anchor">The one-dimensional Ising model </h2>
|
||||
|
||||
<p>
|
||||
In the <b>Least Absolute Shrinkage and Selection Operator</b> (LASSO)-method we get a third cost function.
|
||||
Let us bring back the Ising model again, but now with an additional
|
||||
focus on Ridge and Lasso regression as well. We repeat some of the
|
||||
basic parts of the Ising model and the setup of the training and test
|
||||
data. The one-dimensional Ising model with nearest neighbor
|
||||
interaction, no external field and a constant coupling constant \( J \) is
|
||||
given by
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}.
|
||||
\tag{12}
|
||||
H = -J \sum_{k}^L s_k s_{k + 1},
|
||||
\tag{7}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
where \( s_i \in \{-1, 1\} \) and \( s_{N + 1} = s_1 \). The number of spins in the system is determined by \( L \). For the one-dimensional system there is no phase transition.
|
||||
|
||||
<p>
|
||||
Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from <b>Scikit-Learn</b>.
|
||||
We will look at a system of \( L = 40 \) spins with a coupling constant of \( J = 1 \). To get enough training data we will generate 10000 states with their respective energies.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>clf_lasso <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>Lasso(alpha<span style="color: #666666">=</span>_lambda)<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
J_lasso_sk <span style="color: #666666">=</span> clf_lasso<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J_lasso_sk, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"Lasso from Scikit-learn"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">mpl_toolkits.axes_grid1</span> <span style="color: #008000; font-weight: bold">import</span> make_axes_locatable
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">seaborn</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">sns</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">scipy.linalg</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">scl</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">skl</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">tqdm</span>
|
||||
sns<span style="color: #666666">.</span>set(color_codes<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)
|
||||
cmap_args<span style="color: #666666">=</span><span style="color: #008000">dict</span>(vmin<span style="color: #666666">=-1.</span>, vmax<span style="color: #666666">=1.</span>, cmap<span style="color: #666666">=</span><span style="color: #BA2121">'seismic'</span>)
|
||||
|
||||
L <span style="color: #666666">=</span> <span style="color: #666666">40</span>
|
||||
n <span style="color: #666666">=</span> <span style="color: #008000">int</span>(<span style="color: #666666">1e4</span>)
|
||||
|
||||
spins <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>choice([<span style="color: #666666">-1</span>, <span style="color: #666666">1</span>], size<span style="color: #666666">=</span>(n, L))
|
||||
J <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
energies <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(n)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
energies[i] <span style="color: #666666">=</span> <span style="color: #666666">-</span> J <span style="color: #666666">*</span> np<span style="color: #666666">.</span>dot(spins[i], np<span style="color: #666666">.</span>roll(spins[i], <span style="color: #666666">1</span>))
|
||||
</pre></div>
|
||||
<p>
|
||||
A more general form for the one-dimensional Ising model is
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
H = - \sum_j^L \sum_k^L s_j s_k J_{jk}.
|
||||
\tag{8}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
Here we allow for interactions beyond the nearest neighbors and a more
|
||||
adaptive coupling matrix. This latter expression can be formulated as
|
||||
a matrix-product on the form
|
||||
$$
|
||||
\begin{align}
|
||||
H = X J,
|
||||
\tag{9}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
where \( X_{jk} = s_j s_k \) and \( J \) is the matrix consisting of the
|
||||
elements \( -J_{jk} \). This form of writing the energy fits perfectly
|
||||
with the form utilized in linear regression, viz.
|
||||
$$
|
||||
\begin{align}
|
||||
\boldsymbol{y} = \boldsymbol{X}\boldsymbol{\beta} + \boldsymbol{\epsilon}.
|
||||
\tag{10}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
We organize the data as we did above
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n, L <span style="color: #666666">**</span> <span style="color: #666666">2</span>))
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(n):
|
||||
X[i] <span style="color: #666666">=</span> np<span style="color: #666666">.</span>outer(spins[i], spins[i])<span style="color: #666666">.</span>ravel()
|
||||
y <span style="color: #666666">=</span> energies
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.96</span>)
|
||||
|
||||
X_train_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
(np<span style="color: #666666">.</span>ones(<span style="color: #008000">len</span>(X_train))[:, np<span style="color: #666666">.</span>newaxis], X_train),
|
||||
axis<span style="color: #666666">=1</span>
|
||||
)
|
||||
|
||||
X_test_own <span style="color: #666666">=</span> np<span style="color: #666666">.</span>concatenate(
|
||||
(np<span style="color: #666666">.</span>ones(<span style="color: #008000">len</span>(X_test))[:, np<span style="color: #666666">.</span>newaxis], X_test),
|
||||
axis<span style="color: #666666">=1</span>
|
||||
)
|
||||
</pre></div>
|
||||
<p>
|
||||
We will do all fitting with <b>Scikit-Learn</b>,
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>clf <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>LinearRegression()<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
</pre></div>
|
||||
<p>
|
||||
When extracting the \( J \)-matrix we make sure to remove the intercept
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>J_sk <span style="color: #666666">=</span> clf<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
</pre></div>
|
||||
<p>
|
||||
And then we plot the results
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J_sk, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"LinearRegression from Scikit-learn"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xticks(fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>yticks(fontsize<span style="color: #666666">=18</span>)
|
||||
cb <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>colorbar(im)
|
||||
cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>set_yticklabels(cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>get_yticklabels(), fontsize<span style="color: #666666">=18</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
It is quite striking how LASSO breaks the symmetry of the coupling
|
||||
constant as opposed to ridge and OLS. We get a sparse solution with
|
||||
\( J_{j, j + 1} = -1 \).
|
||||
The results perfectly with our previous discussion where we used our own code.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -470,7 +571,7 @@ constant as opposed to ridge and OLS. We get a sparse solution with
|
||||
<li><a href="._week38-bs027.html">28</a></li>
|
||||
<li><a href="._week38-bs028.html">29</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs020.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,57 +414,38 @@ MathJax.Hub.Config({
|
||||
<a name="part0020"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="performance-as-function-of-the-regularization-parameter" class="anchor">Performance as function of the regularization parameter </h2>
|
||||
<h2 id="ridge-regression" class="anchor">Ridge regression </h2>
|
||||
|
||||
<p>
|
||||
We see how the different models perform for a different set of values for \( \lambda \).
|
||||
Having explored the ordinary least squares we move on to ridge
|
||||
regression. In ridge regression we include a <b>regularizer</b>. This
|
||||
involves a new cost function which leads to a new estimate for the
|
||||
weights \( \boldsymbol{\beta} \). This results in a penalized regression problem. The
|
||||
cost function is given by
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta}.
|
||||
\tag{11}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">5</span>, <span style="color: #666666">10</span>)
|
||||
|
||||
train_errors <span style="color: #666666">=</span> {
|
||||
<span style="color: #BA2121">"ols_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"ridge_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"lasso_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size)
|
||||
}
|
||||
|
||||
test_errors <span style="color: #666666">=</span> {
|
||||
<span style="color: #BA2121">"ols_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"ridge_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"lasso_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size)
|
||||
}
|
||||
|
||||
plot_counter <span style="color: #666666">=</span> <span style="color: #666666">1</span>
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">32</span>, <span style="color: #666666">54</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i, _lambda <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(tqdm<span style="color: #666666">.</span>tqdm(lambdas)):
|
||||
<span style="color: #008000; font-weight: bold">for</span> key, method <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">zip</span>(
|
||||
[<span style="color: #BA2121">"ols_sk"</span>, <span style="color: #BA2121">"ridge_sk"</span>, <span style="color: #BA2121">"lasso_sk"</span>],
|
||||
[skl<span style="color: #666666">.</span>LinearRegression(), skl<span style="color: #666666">.</span>Ridge(alpha<span style="color: #666666">=</span>_lambda), skl<span style="color: #666666">.</span>Lasso(alpha<span style="color: #666666">=</span>_lambda)]
|
||||
):
|
||||
method <span style="color: #666666">=</span> method<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
|
||||
train_errors[key][i] <span style="color: #666666">=</span> method<span style="color: #666666">.</span>score(X_train, y_train)
|
||||
test_errors[key][i] <span style="color: #666666">=</span> method<span style="color: #666666">.</span>score(X_test, y_test)
|
||||
|
||||
omega <span style="color: #666666">=</span> method<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
|
||||
plt<span style="color: #666666">.</span>subplot(<span style="color: #666666">10</span>, <span style="color: #666666">5</span>, plot_counter)
|
||||
plt<span style="color: #666666">.</span>imshow(omega, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r"</span><span style="color: #BB6688; font-weight: bold">%s</span><span style="color: #BA2121">, $\lambda = </span><span style="color: #BB6688; font-weight: bold">%.4f</span><span style="color: #BA2121">$"</span> <span style="color: #666666">%</span> (key, _lambda))
|
||||
plot_counter <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>_lambda <span style="color: #666666">=</span> <span style="color: #666666">0.1</span>
|
||||
clf_ridge <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>Ridge(alpha<span style="color: #666666">=</span>_lambda)<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
J_ridge_sk <span style="color: #666666">=</span> clf_ridge<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J_ridge_sk, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"Ridge from Scikit-learn"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xticks(fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>yticks(fontsize<span style="color: #666666">=18</span>)
|
||||
cb <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>colorbar(im)
|
||||
cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>set_yticklabels(cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>get_yticklabels(), fontsize<span style="color: #666666">=18</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We see that LASSO reaches a good solution for low
|
||||
values of \( \lambda \), but will "wither" when we increase \( \lambda \) too
|
||||
much. Ridge is more stable over a larger range of values for
|
||||
\( \lambda \), but eventually also fades away.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -486,7 +472,7 @@ much. Ridge is more stable over a larger range of values for
|
||||
<li><a href="._week38-bs028.html">29</a></li>
|
||||
<li><a href="._week38-bs029.html">30</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs021.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,54 +414,40 @@ MathJax.Hub.Config({
|
||||
<a name="part0021"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="finding-the-optimal-value-of-lambda" class="anchor">Finding the optimal value of \( \lambda \) </h2>
|
||||
<h2 id="lasso-regression" class="anchor">LASSO regression </h2>
|
||||
|
||||
<p>
|
||||
To determine which value of \( \lambda \) is best we plot the accuracy of
|
||||
the models when predicting the training and the testing set. We expect
|
||||
the accuracy of the training set to be quite good, but if the accuracy
|
||||
of the testing set is much lower this tells us that we might be
|
||||
subject to an overfit model. The ideal scenario is an accuracy on the
|
||||
testing set that is close to the accuracy of the training set.
|
||||
In the <b>Least Absolute Shrinkage and Selection Operator</b> (LASSO)-method we get a third cost function.
|
||||
|
||||
$$
|
||||
\begin{align}
|
||||
C(\boldsymbol{X}, \boldsymbol{\beta}; \lambda) = (\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y})^T(\boldsymbol{X}\boldsymbol{\beta} - \boldsymbol{y}) + \lambda \sqrt{\boldsymbol{\beta}^T\boldsymbol{\beta}}.
|
||||
\tag{12}
|
||||
\end{align}
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finding the extremal point of this cost function is not so straight-forward as in least squares and ridge. We will therefore rely solely on the function ``Lasso`` from <b>Scikit-Learn</b>.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>clf_lasso <span style="color: #666666">=</span> skl<span style="color: #666666">.</span>Lasso(alpha<span style="color: #666666">=</span>_lambda)<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
J_lasso_sk <span style="color: #666666">=</span> clf_lasso<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
im <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>imshow(J_lasso_sk, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">"Lasso from Scikit-learn"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xticks(fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>yticks(fontsize<span style="color: #666666">=18</span>)
|
||||
cb <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>colorbar(im)
|
||||
cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>set_yticklabels(cb<span style="color: #666666">.</span>ax<span style="color: #666666">.</span>get_yticklabels(), fontsize<span style="color: #666666">=18</span>)
|
||||
|
||||
colors <span style="color: #666666">=</span> {
|
||||
<span style="color: #BA2121">"ols_sk"</span>: <span style="color: #BA2121">"r"</span>,
|
||||
<span style="color: #BA2121">"ridge_sk"</span>: <span style="color: #BA2121">"y"</span>,
|
||||
<span style="color: #BA2121">"lasso_sk"</span>: <span style="color: #BA2121">"c"</span>
|
||||
}
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> key <span style="color: #AA22FF; font-weight: bold">in</span> train_errors:
|
||||
plt<span style="color: #666666">.</span>semilogx(
|
||||
lambdas,
|
||||
train_errors[key],
|
||||
colors[key],
|
||||
label<span style="color: #666666">=</span><span style="color: #BA2121">"Train </span><span style="color: #BB6688; font-weight: bold">{0}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(key),
|
||||
linewidth<span style="color: #666666">=4.0</span>
|
||||
)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> key <span style="color: #AA22FF; font-weight: bold">in</span> test_errors:
|
||||
plt<span style="color: #666666">.</span>semilogx(
|
||||
lambdas,
|
||||
test_errors[key],
|
||||
colors[key] <span style="color: #666666">+</span> <span style="color: #BA2121">"--"</span>,
|
||||
label<span style="color: #666666">=</span><span style="color: #BA2121">"Test </span><span style="color: #BB6688; font-weight: bold">{0}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(key),
|
||||
linewidth<span style="color: #666666">=4.0</span>
|
||||
)
|
||||
plt<span style="color: #666666">.</span>legend(loc<span style="color: #666666">=</span><span style="color: #BA2121">"best"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r"$\lambda$"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r"$R^2$"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>tick_params(labelsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
From the above figure we can see that LASSO with \( \lambda = 10^{-2} \)
|
||||
achieves a very good accuracy on the test set. This by far surpasses the
|
||||
other models for all values of \( \lambda \).
|
||||
It is quite striking how LASSO breaks the symmetry of the coupling
|
||||
constant as opposed to ridge and OLS. We get a sparse solution with
|
||||
\( J_{j, j + 1} = -1 \).
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -484,7 +475,7 @@ other models for all values of \( \lambda \).
|
||||
<li><a href="._week38-bs029.html">30</a></li>
|
||||
<li><a href="._week38-bs030.html">31</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs022.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -407,22 +412,58 @@ MathJax.Hub.Config({
|
||||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||||
|
||||
<a name="part0022"></a>
|
||||
<!-- !split -->
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="logistic-regression" class="anchor">Logistic Regression </h2>
|
||||
<h2 id="performance-as-function-of-the-regularization-parameter" class="anchor">Performance as function of the regularization parameter </h2>
|
||||
|
||||
<p>
|
||||
In linear regression our main interest was centered on learning the
|
||||
coefficients of a functional fit (say a polynomial) in order to be
|
||||
able to predict the response of a continuous variable on some unseen
|
||||
data. The fit to the continuous variable \( y_i \) is based on some
|
||||
independent variables \( \hat{x}_i \). Linear regression resulted in
|
||||
analytical expressions for standard ordinary Least Squares or Ridge
|
||||
regression (in terms of matrices to invert) for several quantities,
|
||||
ranging from the variance and thereby the confidence intervals of the
|
||||
parameters \( \hat{\beta} \) to the mean squared error. If we can invert
|
||||
the product of the design matrices, linear regression gives then a
|
||||
simple recipe for fitting our data.
|
||||
We see how the different models perform for a different set of values for \( \lambda \).
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">5</span>, <span style="color: #666666">10</span>)
|
||||
|
||||
train_errors <span style="color: #666666">=</span> {
|
||||
<span style="color: #BA2121">"ols_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"ridge_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"lasso_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size)
|
||||
}
|
||||
|
||||
test_errors <span style="color: #666666">=</span> {
|
||||
<span style="color: #BA2121">"ols_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"ridge_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size),
|
||||
<span style="color: #BA2121">"lasso_sk"</span>: np<span style="color: #666666">.</span>zeros(lambdas<span style="color: #666666">.</span>size)
|
||||
}
|
||||
|
||||
plot_counter <span style="color: #666666">=</span> <span style="color: #666666">1</span>
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">32</span>, <span style="color: #666666">54</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i, _lambda <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">enumerate</span>(tqdm<span style="color: #666666">.</span>tqdm(lambdas)):
|
||||
<span style="color: #008000; font-weight: bold">for</span> key, method <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">zip</span>(
|
||||
[<span style="color: #BA2121">"ols_sk"</span>, <span style="color: #BA2121">"ridge_sk"</span>, <span style="color: #BA2121">"lasso_sk"</span>],
|
||||
[skl<span style="color: #666666">.</span>LinearRegression(), skl<span style="color: #666666">.</span>Ridge(alpha<span style="color: #666666">=</span>_lambda), skl<span style="color: #666666">.</span>Lasso(alpha<span style="color: #666666">=</span>_lambda)]
|
||||
):
|
||||
method <span style="color: #666666">=</span> method<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
|
||||
train_errors[key][i] <span style="color: #666666">=</span> method<span style="color: #666666">.</span>score(X_train, y_train)
|
||||
test_errors[key][i] <span style="color: #666666">=</span> method<span style="color: #666666">.</span>score(X_test, y_test)
|
||||
|
||||
omega <span style="color: #666666">=</span> method<span style="color: #666666">.</span>coef_<span style="color: #666666">.</span>reshape(L, L)
|
||||
|
||||
plt<span style="color: #666666">.</span>subplot(<span style="color: #666666">10</span>, <span style="color: #666666">5</span>, plot_counter)
|
||||
plt<span style="color: #666666">.</span>imshow(omega, <span style="color: #666666">**</span>cmap_args)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r"</span><span style="color: #BB6688; font-weight: bold">%s</span><span style="color: #BA2121">, $\lambda = </span><span style="color: #BB6688; font-weight: bold">%.4f</span><span style="color: #BA2121">$"</span> <span style="color: #666666">%</span> (key, _lambda))
|
||||
plot_counter <span style="color: #666666">+=</span> <span style="color: #666666">1</span>
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We see that LASSO reaches a good solution for low
|
||||
values of \( \lambda \), but will "wither" when we increase \( \lambda \) too
|
||||
much. Ridge is more stable over a larger range of values for
|
||||
\( \lambda \), but eventually also fades away.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -450,7 +491,7 @@ simple recipe for fitting our data.
|
||||
<li><a href="._week38-bs030.html">31</a></li>
|
||||
<li><a href="._week38-bs031.html">32</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs023.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -407,27 +412,56 @@ MathJax.Hub.Config({
|
||||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||||
|
||||
<a name="part0023"></a>
|
||||
<!-- !split -->
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="classification-problems" class="anchor">Classification problems </h2>
|
||||
<h2 id="finding-the-optimal-value-of-lambda" class="anchor">Finding the optimal value of \( \lambda \) </h2>
|
||||
|
||||
<p>
|
||||
Classification problems, however, are concerned with outcomes taking
|
||||
the form of discrete variables (i.e. categories). We may for example,
|
||||
on the basis of DNA sequencing for a number of patients, like to find
|
||||
out which mutations are important for a certain disease; or based on
|
||||
scans of various patients' brains, figure out if there is a tumor or
|
||||
not; or given a specific physical system, we'd like to identify its
|
||||
state, say whether it is an ordered or disordered system (typical
|
||||
situation in solid state physics); or classify the status of a
|
||||
patient, whether she/he has a stroke or not and many other similar
|
||||
situations.
|
||||
To determine which value of \( \lambda \) is best we plot the accuracy of
|
||||
the models when predicting the training and the testing set. We expect
|
||||
the accuracy of the training set to be quite good, but if the accuracy
|
||||
of the testing set is much lower this tells us that we might be
|
||||
subject to an overfit model. The ideal scenario is an accuracy on the
|
||||
testing set that is close to the accuracy of the training set.
|
||||
|
||||
<p>
|
||||
The most common situation we encounter when we apply logistic
|
||||
regression is that of two possible outcomes, normally denoted as a
|
||||
binary outcome, true or false, positive or negative, success or
|
||||
failure etc.
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">20</span>, <span style="color: #666666">14</span>))
|
||||
|
||||
colors <span style="color: #666666">=</span> {
|
||||
<span style="color: #BA2121">"ols_sk"</span>: <span style="color: #BA2121">"r"</span>,
|
||||
<span style="color: #BA2121">"ridge_sk"</span>: <span style="color: #BA2121">"y"</span>,
|
||||
<span style="color: #BA2121">"lasso_sk"</span>: <span style="color: #BA2121">"c"</span>
|
||||
}
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> key <span style="color: #AA22FF; font-weight: bold">in</span> train_errors:
|
||||
plt<span style="color: #666666">.</span>semilogx(
|
||||
lambdas,
|
||||
train_errors[key],
|
||||
colors[key],
|
||||
label<span style="color: #666666">=</span><span style="color: #BA2121">"Train </span><span style="color: #BB6688; font-weight: bold">{0}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(key),
|
||||
linewidth<span style="color: #666666">=4.0</span>
|
||||
)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> key <span style="color: #AA22FF; font-weight: bold">in</span> test_errors:
|
||||
plt<span style="color: #666666">.</span>semilogx(
|
||||
lambdas,
|
||||
test_errors[key],
|
||||
colors[key] <span style="color: #666666">+</span> <span style="color: #BA2121">"--"</span>,
|
||||
label<span style="color: #666666">=</span><span style="color: #BA2121">"Test </span><span style="color: #BB6688; font-weight: bold">{0}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(key),
|
||||
linewidth<span style="color: #666666">=4.0</span>
|
||||
)
|
||||
plt<span style="color: #666666">.</span>legend(loc<span style="color: #666666">=</span><span style="color: #BA2121">"best"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r"$\lambda$"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r"$R^2$"</span>, fontsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>tick_params(labelsize<span style="color: #666666">=18</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
From the above figure we can see that LASSO with \( \lambda = 10^{-2} \)
|
||||
achieves a very good accuracy on the test set. This by far surpasses the
|
||||
other models for all values of \( \lambda \).
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -455,7 +489,7 @@ failure etc.
|
||||
<li><a href="._week38-bs031.html">32</a></li>
|
||||
<li><a href="._week38-bs032.html">33</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs024.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -407,25 +412,22 @@ MathJax.Hub.Config({
|
||||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||||
|
||||
<a name="part0024"></a>
|
||||
<!-- !split -->
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="optimization-and-deep-learning" class="anchor">Optimization and Deep learning </h2>
|
||||
<h2 id="logistic-regression" class="anchor">Logistic Regression </h2>
|
||||
|
||||
<p>
|
||||
Logistic regression will also serve as our stepping stone towards
|
||||
neural network algorithms and supervised deep learning. For logistic
|
||||
learning, the minimization of the cost function leads to a non-linear
|
||||
equation in the parameters \( \hat{\beta} \). The optimization of the
|
||||
problem calls therefore for minimization algorithms. This forms the
|
||||
bottle neck of all machine learning algorithms, namely how to find
|
||||
reliable minima of a multi-variable function. This leads us to the
|
||||
family of gradient descent methods. The latter are the working horses
|
||||
of basically all modern machine learning algorithms.
|
||||
|
||||
<p>
|
||||
We note also that many of the topics discussed here on logistic
|
||||
regression are also commonly used in modern supervised Deep Learning
|
||||
models, as we will see later.
|
||||
In linear regression our main interest was centered on learning the
|
||||
coefficients of a functional fit (say a polynomial) in order to be
|
||||
able to predict the response of a continuous variable on some unseen
|
||||
data. The fit to the continuous variable \( y_i \) is based on some
|
||||
independent variables \( \hat{x}_i \). Linear regression resulted in
|
||||
analytical expressions for standard ordinary Least Squares or Ridge
|
||||
regression (in terms of matrices to invert) for several quantities,
|
||||
ranging from the variance and thereby the confidence intervals of the
|
||||
parameters \( \hat{\beta} \) to the mean squared error. If we can invert
|
||||
the product of the design matrices, linear regression gives then a
|
||||
simple recipe for fitting our data.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -453,7 +455,7 @@ models, as we will see later.
|
||||
<li><a href="._week38-bs032.html">33</a></li>
|
||||
<li><a href="._week38-bs033.html">34</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs025.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,29 +414,25 @@ MathJax.Hub.Config({
|
||||
<a name="part0025"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="basics" class="anchor">Basics </h2>
|
||||
<h2 id="classification-problems" class="anchor">Classification problems </h2>
|
||||
|
||||
<p>
|
||||
We consider the case where the dependent variables, also called the
|
||||
responses or the outcomes, \( y_i \) are discrete and only take values
|
||||
from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).
|
||||
Classification problems, however, are concerned with outcomes taking
|
||||
the form of discrete variables (i.e. categories). We may for example,
|
||||
on the basis of DNA sequencing for a number of patients, like to find
|
||||
out which mutations are important for a certain disease; or based on
|
||||
scans of various patients' brains, figure out if there is a tumor or
|
||||
not; or given a specific physical system, we'd like to identify its
|
||||
state, say whether it is an ordered or disordered system (typical
|
||||
situation in solid state physics); or classify the status of a
|
||||
patient, whether she/he has a stroke or not and many other similar
|
||||
situations.
|
||||
|
||||
<p>
|
||||
The goal is to predict the
|
||||
output classes from the design matrix \( \hat{X}\in\mathbb{R}^{n\times p} \)
|
||||
made of \( n \) samples, each of which carries \( p \) features or predictors. The
|
||||
primary goal is to identify the classes to which new unseen samples
|
||||
belong.
|
||||
|
||||
<p>
|
||||
Let us specialize to the case of two classes only, with outputs
|
||||
\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a
|
||||
credit card user that could default or not on her/his credit card
|
||||
debt. That is
|
||||
|
||||
$$
|
||||
y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}.
|
||||
$$
|
||||
The most common situation we encounter when we apply logistic
|
||||
regression is that of two possible outcomes, normally denoted as a
|
||||
binary outcome, true or false, positive or negative, success or
|
||||
failure etc.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -459,7 +460,7 @@ $$
|
||||
<li><a href="._week38-bs033.html">34</a></li>
|
||||
<li><a href="._week38-bs034.html">35</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs026.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,26 +414,23 @@ MathJax.Hub.Config({
|
||||
<a name="part0026"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="linear-classifier" class="anchor">Linear classifier </h2>
|
||||
<h2 id="optimization-and-deep-learning" class="anchor">Optimization and Deep learning </h2>
|
||||
|
||||
<p>
|
||||
Before moving to the logistic model, let us try to use our linear
|
||||
regression model to classify these two outcomes. We could for example
|
||||
fit a linear model to the default case if \( y_i > 0.5 \) and the no
|
||||
default case \( y_i \leq 0.5 \).
|
||||
Logistic regression will also serve as our stepping stone towards
|
||||
neural network algorithms and supervised deep learning. For logistic
|
||||
learning, the minimization of the cost function leads to a non-linear
|
||||
equation in the parameters \( \hat{\beta} \). The optimization of the
|
||||
problem calls therefore for minimization algorithms. This forms the
|
||||
bottle neck of all machine learning algorithms, namely how to find
|
||||
reliable minima of a multi-variable function. This leads us to the
|
||||
family of gradient descent methods. The latter are the working horses
|
||||
of basically all modern machine learning algorithms.
|
||||
|
||||
<p>
|
||||
We would then have our
|
||||
weighted linear combination, namely
|
||||
$$
|
||||
\begin{equation}
|
||||
\hat{y} = \hat{X}^T\hat{\beta} + \hat{\epsilon},
|
||||
\tag{13}
|
||||
\end{equation}
|
||||
$$
|
||||
|
||||
where \( \hat{y} \) is a vector representing the possible outcomes, \( \hat{X} \) is our
|
||||
\( n\times p \) design matrix and \( \hat{\beta} \) represents our estimators/predictors.
|
||||
We note also that many of the topics discussed here on logistic
|
||||
regression are also commonly used in modern supervised Deep Learning
|
||||
models, as we will see later.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -456,7 +458,7 @@ where \( \hat{y} \) is a vector representing the possible outcomes, \( \hat{X} \
|
||||
<li><a href="._week38-bs034.html">35</a></li>
|
||||
<li><a href="._week38-bs035.html">36</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs027.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -407,26 +412,31 @@ MathJax.Hub.Config({
|
||||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||||
|
||||
<a name="part0027"></a>
|
||||
<!-- !split -->
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="some-selected-properties" class="anchor">Some selected properties </h2>
|
||||
<h2 id="basics" class="anchor">Basics </h2>
|
||||
|
||||
<p>
|
||||
The main problem with our function is that it takes values on the
|
||||
entire real axis. In the case of logistic regression, however, the
|
||||
labels \( y_i \) are discrete variables. A typical example is the credit
|
||||
card data discussed below here, where we can set the state of
|
||||
defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons
|
||||
in the data set (see the full example below).
|
||||
We consider the case where the dependent variables, also called the
|
||||
responses or the outcomes, \( y_i \) are discrete and only take values
|
||||
from \( k=0,\dots,K-1 \) (i.e. \( K \) classes).
|
||||
|
||||
<p>
|
||||
One simple way to get a discrete output is to have sign
|
||||
functions that map the output of a linear regressor to values \( \{0,1\} \),
|
||||
\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise.
|
||||
We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning
|
||||
literature. This model is extremely simple. However, in many cases it is more
|
||||
favorable to use a ``soft" classifier that outputs
|
||||
the probability of a given category. This leads us to the logistic function.
|
||||
The goal is to predict the
|
||||
output classes from the design matrix \( \hat{X}\in\mathbb{R}^{n\times p} \)
|
||||
made of \( n \) samples, each of which carries \( p \) features or predictors. The
|
||||
primary goal is to identify the classes to which new unseen samples
|
||||
belong.
|
||||
|
||||
<p>
|
||||
Let us specialize to the case of two classes only, with outputs
|
||||
\( y_i=0 \) and \( y_i=1 \). Our outcomes could represent the status of a
|
||||
credit card user that could default or not on her/his credit card
|
||||
debt. That is
|
||||
|
||||
$$
|
||||
y_i = \begin{bmatrix} 0 & \mathrm{no}\\ 1 & \mathrm{yes} \end{bmatrix}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -454,7 +464,7 @@ the probability of a given category. This leads us to the logistic function.
|
||||
<li><a href="._week38-bs035.html">36</a></li>
|
||||
<li><a href="._week38-bs036.html">37</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs028.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,69 +414,27 @@ MathJax.Hub.Config({
|
||||
<a name="part0028"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="simple-example" class="anchor">Simple example </h2>
|
||||
<h2 id="linear-classifier" class="anchor">Linear classifier </h2>
|
||||
|
||||
<p>
|
||||
The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.
|
||||
Before moving to the logistic model, let us try to use our linear
|
||||
regression model to classify these two outcomes. We could for example
|
||||
fit a linear model to the default case if \( y_i > 0.5 \) and the no
|
||||
default case \( y_i \leq 0.5 \).
|
||||
|
||||
<p>
|
||||
We would then have our
|
||||
weighted linear combination, namely
|
||||
$$
|
||||
\begin{equation}
|
||||
\hat{y} = \hat{X}^T\hat{\beta} + \hat{\epsilon},
|
||||
\tag{13}
|
||||
\end{equation}
|
||||
$$
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #408080; font-style: italic"># Common imports</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">os</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression, Ridge, Lasso
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.utils</span> <span style="color: #008000; font-weight: bold">import</span> resample
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.metrics</span> <span style="color: #008000; font-weight: bold">import</span> mean_squared_error
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">IPython.display</span> <span style="color: #008000; font-weight: bold">import</span> display
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">pylab</span> <span style="color: #008000; font-weight: bold">import</span> plt, mpl
|
||||
plt<span style="color: #666666">.</span>style<span style="color: #666666">.</span>use(<span style="color: #BA2121">'seaborn'</span>)
|
||||
mpl<span style="color: #666666">.</span>rcParams[<span style="color: #BA2121">'font.family'</span>] <span style="color: #666666">=</span> <span style="color: #BA2121">'serif'</span>
|
||||
where \( \hat{y} \) is a vector representing the possible outcomes, \( \hat{X} \) is our
|
||||
\( n\times p \) design matrix and \( \hat{\beta} \) represents our estimators/predictors.
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Where to save the figures and data files</span>
|
||||
PROJECT_ROOT_DIR <span style="color: #666666">=</span> <span style="color: #BA2121">"Results"</span>
|
||||
FIGURE_ID <span style="color: #666666">=</span> <span style="color: #BA2121">"Results/FigureFiles"</span>
|
||||
DATA_ID <span style="color: #666666">=</span> <span style="color: #BA2121">"DataFiles/"</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">if</span> <span style="color: #AA22FF; font-weight: bold">not</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>exists(PROJECT_ROOT_DIR):
|
||||
os<span style="color: #666666">.</span>mkdir(PROJECT_ROOT_DIR)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">if</span> <span style="color: #AA22FF; font-weight: bold">not</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>exists(FIGURE_ID):
|
||||
os<span style="color: #666666">.</span>makedirs(FIGURE_ID)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">if</span> <span style="color: #AA22FF; font-weight: bold">not</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>exists(DATA_ID):
|
||||
os<span style="color: #666666">.</span>makedirs(DATA_ID)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">image_path</span>(fig_id):
|
||||
<span style="color: #008000; font-weight: bold">return</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>join(FIGURE_ID, fig_id)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">data_path</span>(dat_id):
|
||||
<span style="color: #008000; font-weight: bold">return</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>join(DATA_ID, dat_id)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">save_fig</span>(fig_id):
|
||||
plt<span style="color: #666666">.</span>savefig(image_path(fig_id) <span style="color: #666666">+</span> <span style="color: #BA2121">".png"</span>, <span style="color: #008000">format</span><span style="color: #666666">=</span><span style="color: #BA2121">'png'</span>)
|
||||
|
||||
infile <span style="color: #666666">=</span> <span style="color: #008000">open</span>(data_path(<span style="color: #BA2121">"chddata.csv"</span>),<span style="color: #BA2121">'r'</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Read the chd data as csv file and organize the data into arrays with age group, age, and chd</span>
|
||||
chd <span style="color: #666666">=</span> pd<span style="color: #666666">.</span>read_csv(infile, names<span style="color: #666666">=</span>(<span style="color: #BA2121">'ID'</span>, <span style="color: #BA2121">'Age'</span>, <span style="color: #BA2121">'Agegroup'</span>, <span style="color: #BA2121">'CHD'</span>))
|
||||
chd<span style="color: #666666">.</span>columns <span style="color: #666666">=</span> [<span style="color: #BA2121">'ID'</span>, <span style="color: #BA2121">'Age'</span>, <span style="color: #BA2121">'Agegroup'</span>, <span style="color: #BA2121">'CHD'</span>]
|
||||
output <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'CHD'</span>]
|
||||
age <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'Age'</span>]
|
||||
agegroup <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'Agegroup'</span>]
|
||||
numberID <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'ID'</span>]
|
||||
display(chd)
|
||||
|
||||
plt<span style="color: #666666">.</span>scatter(age, output, marker<span style="color: #666666">=</span><span style="color: #BA2121">'o'</span>)
|
||||
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">18</span>,<span style="color: #666666">70.0</span>,<span style="color: #666666">-0.1</span>, <span style="color: #666666">1.2</span>])
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'Age'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'CHD'</span>)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Age distribution and Coronary heart disease'</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -498,7 +461,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs036.html">37</a></li>
|
||||
<li><a href="._week38-bs037.html">38</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs029.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,41 +414,24 @@ MathJax.Hub.Config({
|
||||
<a name="part0029"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="plotting-the-mean-value-for-each-group" class="anchor">Plotting the mean value for each group </h2>
|
||||
<h2 id="some-selected-properties" class="anchor">Some selected properties </h2>
|
||||
|
||||
<p>
|
||||
What we could attempt however is to plot the mean value for each group.
|
||||
The main problem with our function is that it takes values on the
|
||||
entire real axis. In the case of logistic regression, however, the
|
||||
labels \( y_i \) are discrete variables. A typical example is the credit
|
||||
card data discussed below here, where we can set the state of
|
||||
defaulting the debt to \( y_i=1 \) and not to \( y_i=0 \) for one the persons
|
||||
in the data set (see the full example below).
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>agegroupmean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">0.1</span>, <span style="color: #666666">0.133</span>, <span style="color: #666666">0.250</span>, <span style="color: #666666">0.333</span>, <span style="color: #666666">0.462</span>, <span style="color: #666666">0.625</span>, <span style="color: #666666">0.765</span>, <span style="color: #666666">0.800</span>])
|
||||
group <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">1</span>, <span style="color: #666666">2</span>, <span style="color: #666666">3</span>, <span style="color: #666666">4</span>, <span style="color: #666666">5</span>, <span style="color: #666666">6</span>, <span style="color: #666666">7</span>, <span style="color: #666666">8</span>])
|
||||
plt<span style="color: #666666">.</span>plot(group, agegroupmean, <span style="color: #BA2121">"r-"</span>)
|
||||
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">0</span>,<span style="color: #666666">9</span>,<span style="color: #666666">0</span>, <span style="color: #666666">1.0</span>])
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'Age group'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'CHD mean values'</span>)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Mean values for each age group'</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \).
|
||||
In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model
|
||||
$$
|
||||
f(y_i\vert x_i)=\beta_0+\beta_1 x_i.
|
||||
$$
|
||||
|
||||
<p>
|
||||
This expression implies however that \( f(y_i\vert x_i) \) could take any
|
||||
value from minus infinity to plus infinity. If we however let
|
||||
\( f(y\vert y) \) be represented by the mean value, the above example
|
||||
shows us that we can constrain the function to take values between
|
||||
zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking
|
||||
at our last curve we see also that it has an S-shaped form. This leads
|
||||
us to a very popular model for the function \( f \), namely the so-called
|
||||
Sigmoid function or logistic model. We will consider this function as
|
||||
representing the probability for finding a value of \( y_i \) with a given
|
||||
\( x_i \).
|
||||
One simple way to get a discrete output is to have sign
|
||||
functions that map the output of a linear regressor to values \( \{0,1\} \),
|
||||
\( f(s_i)=sign(s_i)=1 \) if \( s_i\ge 0 \) and 0 if otherwise.
|
||||
We will encounter this model in our first demonstration of neural networks. Historically it is called the ``perceptron" model in the machine learning
|
||||
literature. This model is extremely simple. However, in many cases it is more
|
||||
favorable to use a ``soft" classifier that outputs
|
||||
the probability of a given category. This leads us to the logistic function.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -471,7 +459,7 @@ representing the probability for finding a value of \( y_i \) with a given
|
||||
<li><a href="._week38-bs037.html">38</a></li>
|
||||
<li><a href="._week38-bs038.html">39</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs030.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,26 +414,69 @@ MathJax.Hub.Config({
|
||||
<a name="part0030"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="the-logistic-function" class="anchor">The logistic function </h2>
|
||||
<h2 id="simple-example" class="anchor">Simple example </h2>
|
||||
|
||||
<p>
|
||||
Another widely studied model, is the so-called
|
||||
perceptron model, which is an example of a "hard classification" model. We
|
||||
will encounter this model when we discuss neural networks as
|
||||
well. Each datapoint is deterministically assigned to a category (i.e
|
||||
\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft"
|
||||
classifier that outputs the probability of a given category rather
|
||||
than a single value. For example, given \( x_i \), the classifier
|
||||
outputs the probability of being in a category \( k \). Logistic regression
|
||||
is the most common example of a so-called soft classifier. In logistic
|
||||
regression, the probability that a data point \( x_i \)
|
||||
belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,
|
||||
$$
|
||||
p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}.
|
||||
$$
|
||||
The following example on data for coronary heart disease (CHD) as function of age may serve as an illustration. In the code here we read and plot whether a person has had CHD (output = 1) or not (output = 0). This ouput is plotted the person's against age. Clearly, the figure shows that attempting to make a standard linear regression fit may not be very meaningful.
|
||||
|
||||
Note that \( 1-p(t)= p(-t) \).
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #408080; font-style: italic"># Common imports</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">os</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LinearRegression, Ridge, Lasso
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.utils</span> <span style="color: #008000; font-weight: bold">import</span> resample
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.metrics</span> <span style="color: #008000; font-weight: bold">import</span> mean_squared_error
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">IPython.display</span> <span style="color: #008000; font-weight: bold">import</span> display
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">pylab</span> <span style="color: #008000; font-weight: bold">import</span> plt, mpl
|
||||
plt<span style="color: #666666">.</span>style<span style="color: #666666">.</span>use(<span style="color: #BA2121">'seaborn'</span>)
|
||||
mpl<span style="color: #666666">.</span>rcParams[<span style="color: #BA2121">'font.family'</span>] <span style="color: #666666">=</span> <span style="color: #BA2121">'serif'</span>
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Where to save the figures and data files</span>
|
||||
PROJECT_ROOT_DIR <span style="color: #666666">=</span> <span style="color: #BA2121">"Results"</span>
|
||||
FIGURE_ID <span style="color: #666666">=</span> <span style="color: #BA2121">"Results/FigureFiles"</span>
|
||||
DATA_ID <span style="color: #666666">=</span> <span style="color: #BA2121">"DataFiles/"</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">if</span> <span style="color: #AA22FF; font-weight: bold">not</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>exists(PROJECT_ROOT_DIR):
|
||||
os<span style="color: #666666">.</span>mkdir(PROJECT_ROOT_DIR)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">if</span> <span style="color: #AA22FF; font-weight: bold">not</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>exists(FIGURE_ID):
|
||||
os<span style="color: #666666">.</span>makedirs(FIGURE_ID)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">if</span> <span style="color: #AA22FF; font-weight: bold">not</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>exists(DATA_ID):
|
||||
os<span style="color: #666666">.</span>makedirs(DATA_ID)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">image_path</span>(fig_id):
|
||||
<span style="color: #008000; font-weight: bold">return</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>join(FIGURE_ID, fig_id)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">data_path</span>(dat_id):
|
||||
<span style="color: #008000; font-weight: bold">return</span> os<span style="color: #666666">.</span>path<span style="color: #666666">.</span>join(DATA_ID, dat_id)
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">save_fig</span>(fig_id):
|
||||
plt<span style="color: #666666">.</span>savefig(image_path(fig_id) <span style="color: #666666">+</span> <span style="color: #BA2121">".png"</span>, <span style="color: #008000">format</span><span style="color: #666666">=</span><span style="color: #BA2121">'png'</span>)
|
||||
|
||||
infile <span style="color: #666666">=</span> <span style="color: #008000">open</span>(data_path(<span style="color: #BA2121">"chddata.csv"</span>),<span style="color: #BA2121">'r'</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Read the chd data as csv file and organize the data into arrays with age group, age, and chd</span>
|
||||
chd <span style="color: #666666">=</span> pd<span style="color: #666666">.</span>read_csv(infile, names<span style="color: #666666">=</span>(<span style="color: #BA2121">'ID'</span>, <span style="color: #BA2121">'Age'</span>, <span style="color: #BA2121">'Agegroup'</span>, <span style="color: #BA2121">'CHD'</span>))
|
||||
chd<span style="color: #666666">.</span>columns <span style="color: #666666">=</span> [<span style="color: #BA2121">'ID'</span>, <span style="color: #BA2121">'Age'</span>, <span style="color: #BA2121">'Agegroup'</span>, <span style="color: #BA2121">'CHD'</span>]
|
||||
output <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'CHD'</span>]
|
||||
age <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'Age'</span>]
|
||||
agegroup <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'Agegroup'</span>]
|
||||
numberID <span style="color: #666666">=</span> chd[<span style="color: #BA2121">'ID'</span>]
|
||||
display(chd)
|
||||
|
||||
plt<span style="color: #666666">.</span>scatter(age, output, marker<span style="color: #666666">=</span><span style="color: #BA2121">'o'</span>)
|
||||
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">18</span>,<span style="color: #666666">70.0</span>,<span style="color: #666666">-0.1</span>, <span style="color: #666666">1.2</span>])
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'Age'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'CHD'</span>)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Age distribution and Coronary heart disease'</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -455,7 +503,7 @@ Note that \( 1-p(t)= p(-t) \).
|
||||
<li><a href="._week38-bs038.html">39</a></li>
|
||||
<li><a href="._week38-bs039.html">40</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs031.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,69 +414,42 @@ MathJax.Hub.Config({
|
||||
<a name="part0031"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" class="anchor">Examples of likelihood functions used in logistic regression and nueral networks </h2>
|
||||
<h2 id="plotting-the-mean-value-for-each-group" class="anchor">Plotting the mean value for each group </h2>
|
||||
|
||||
<p>
|
||||
The following code plots the logistic function, the step function and other functions we will encounter from here and on.
|
||||
What we could attempt however is to plot the mean value for each group.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #BA2121; font-style: italic">"""The sigmoid function (or the logistic curve) is a</span>
|
||||
<span style="color: #BA2121; font-style: italic">function that takes any real number, z, and outputs a number (0,1).</span>
|
||||
<span style="color: #BA2121; font-style: italic">It is useful in neural networks for assigning weights on a relative scale.</span>
|
||||
<span style="color: #BA2121; font-style: italic">The value z is the weighted sum of parameters involved in the learning algorithm."""</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">math</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">mt</span>
|
||||
|
||||
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">.1</span>)
|
||||
sigma_fn <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>vectorize(<span style="color: #008000; font-weight: bold">lambda</span> z: <span style="color: #666666">1/</span>(<span style="color: #666666">1+</span>numpy<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>z)))
|
||||
sigma <span style="color: #666666">=</span> sigma_fn(z)
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
||||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
||||
ax<span style="color: #666666">.</span>plot(z, sigma)
|
||||
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-0.1</span>, <span style="color: #666666">1.1</span>])
|
||||
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-5</span>,<span style="color: #666666">5</span>])
|
||||
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
||||
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
||||
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'sigmoid function'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
|
||||
<span style="color: #BA2121; font-style: italic">"""Step Function"""</span>
|
||||
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">.02</span>)
|
||||
step_fn <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>vectorize(<span style="color: #008000; font-weight: bold">lambda</span> z: <span style="color: #666666">1.0</span> <span style="color: #008000; font-weight: bold">if</span> z <span style="color: #666666">>=</span> <span style="color: #666666">0.0</span> <span style="color: #008000; font-weight: bold">else</span> <span style="color: #666666">0.0</span>)
|
||||
step <span style="color: #666666">=</span> step_fn(z)
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
||||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
||||
ax<span style="color: #666666">.</span>plot(z, step)
|
||||
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-0.5</span>, <span style="color: #666666">1.5</span>])
|
||||
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-5</span>,<span style="color: #666666">5</span>])
|
||||
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
||||
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
||||
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'step function'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
|
||||
<span style="color: #BA2121; font-style: italic">"""tanh Function"""</span>
|
||||
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-2*</span>mt<span style="color: #666666">.</span>pi, <span style="color: #666666">2*</span>mt<span style="color: #666666">.</span>pi, <span style="color: #666666">0.1</span>)
|
||||
t <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>tanh(z)
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
||||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
||||
ax<span style="color: #666666">.</span>plot(z, t)
|
||||
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-1.0</span>, <span style="color: #666666">1.0</span>])
|
||||
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-2*</span>mt<span style="color: #666666">.</span>pi,<span style="color: #666666">2*</span>mt<span style="color: #666666">.</span>pi])
|
||||
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
||||
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
||||
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'tanh function'</span>)
|
||||
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>agegroupmean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">0.1</span>, <span style="color: #666666">0.133</span>, <span style="color: #666666">0.250</span>, <span style="color: #666666">0.333</span>, <span style="color: #666666">0.462</span>, <span style="color: #666666">0.625</span>, <span style="color: #666666">0.765</span>, <span style="color: #666666">0.800</span>])
|
||||
group <span style="color: #666666">=</span> np<span style="color: #666666">.</span>array([<span style="color: #666666">1</span>, <span style="color: #666666">2</span>, <span style="color: #666666">3</span>, <span style="color: #666666">4</span>, <span style="color: #666666">5</span>, <span style="color: #666666">6</span>, <span style="color: #666666">7</span>, <span style="color: #666666">8</span>])
|
||||
plt<span style="color: #666666">.</span>plot(group, agegroupmean, <span style="color: #BA2121">"r-"</span>)
|
||||
plt<span style="color: #666666">.</span>axis([<span style="color: #666666">0</span>,<span style="color: #666666">9</span>,<span style="color: #666666">0</span>, <span style="color: #666666">1.0</span>])
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">r'Age group'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">r'CHD mean values'</span>)
|
||||
plt<span style="color: #666666">.</span>title(<span style="color: #BA2121">r'Mean values for each age group'</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
We are now trying to find a function \( f(y\vert x) \), that is a function which gives us an expected value for the output \( y \) with a given input \( x \).
|
||||
In standard linear regression with a linear dependence on \( x \), we would write this in terms of our model
|
||||
$$
|
||||
f(y_i\vert x_i)=\beta_0+\beta_1 x_i.
|
||||
$$
|
||||
|
||||
<p>
|
||||
This expression implies however that \( f(y_i\vert x_i) \) could take any
|
||||
value from minus infinity to plus infinity. If we however let
|
||||
\( f(y\vert y) \) be represented by the mean value, the above example
|
||||
shows us that we can constrain the function to take values between
|
||||
zero and one, that is we have \( 0 \le f(y_i\vert x_i) \le 1 \). Looking
|
||||
at our last curve we see also that it has an S-shaped form. This leads
|
||||
us to a very popular model for the function \( f \), namely the so-called
|
||||
Sigmoid function or logistic model. We will consider this function as
|
||||
representing the probability for finding a value of \( y_i \) with a given
|
||||
\( x_i \).
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -498,7 +476,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs039.html">40</a></li>
|
||||
<li><a href="._week38-bs040.html">41</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs032.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,24 +414,25 @@ MathJax.Hub.Config({
|
||||
<a name="part0032"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="two-parameters" class="anchor">Two parameters </h2>
|
||||
<h2 id="the-logistic-function" class="anchor">The logistic function </h2>
|
||||
|
||||
<p>
|
||||
We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities
|
||||
Another widely studied model, is the so-called
|
||||
perceptron model, which is an example of a "hard classification" model. We
|
||||
will encounter this model when we discuss neural networks as
|
||||
well. Each datapoint is deterministically assigned to a category (i.e
|
||||
\( y_i=0 \) or \( y_i=1 \)). In many cases, and the coronary heart disease data forms one of many such examples, it is favorable to have a "soft"
|
||||
classifier that outputs the probability of a given category rather
|
||||
than a single value. For example, given \( x_i \), the classifier
|
||||
outputs the probability of being in a category \( k \). Logistic regression
|
||||
is the most common example of a so-called soft classifier. In logistic
|
||||
regression, the probability that a data point \( x_i \)
|
||||
belongs to a category \( y_i=\{0,1\} \) is given by the so-called logit function (or Sigmoid) which is meant to represent the likelihood for a given event,
|
||||
$$
|
||||
\begin{align*}
|
||||
p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
|
||||
p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}),
|
||||
\end{align*}
|
||||
p(t) = \frac{1}{1+\mathrm \exp{-t}}=\frac{\exp{t}}{1+\mathrm \exp{t}}.
|
||||
$$
|
||||
|
||||
where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
|
||||
|
||||
<p>
|
||||
Note that we used
|
||||
$$
|
||||
p(y_i=0\vert x_i, \hat{\beta}) = 1-p(y_i=1\vert x_i, \hat{\beta}).
|
||||
$$
|
||||
Note that \( 1-p(t)= p(-t) \).
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -454,7 +460,7 @@ $$
|
||||
<li><a href="._week38-bs040.html">41</a></li>
|
||||
<li><a href="._week38-bs041.html">42</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs033.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -407,28 +412,71 @@ MathJax.Hub.Config({
|
||||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||||
|
||||
<a name="part0033"></a>
|
||||
<!-- !split -->
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="maximum-likelihood" class="anchor">Maximum likelihood </h2>
|
||||
<h2 id="examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" class="anchor">Examples of likelihood functions used in logistic regression and nueral networks </h2>
|
||||
|
||||
<p>
|
||||
In order to define the total likelihood for all possible outcomes from a
|
||||
dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels
|
||||
\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called <a href="https://en.wikipedia.org/wiki/Maximum_likelihood_estimation" target="_self">Maximum Likelihood Estimation</a> (MLE) principle.
|
||||
We aim thus at maximizing
|
||||
the probability of seeing the observed data. We can then approximate the
|
||||
likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is
|
||||
$$
|
||||
\begin{align*}
|
||||
P(\mathcal{D}|\hat{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\hat{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\hat{\beta}))\right]^{1-y_i}\nonumber \\
|
||||
\end{align*}
|
||||
$$
|
||||
The following code plots the logistic function, the step function and other functions we will encounter from here and on.
|
||||
|
||||
from which we obtain the log-likelihood and our <b>cost/loss</b> function
|
||||
$$
|
||||
\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\hat{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\hat{\beta}))\right]\right).
|
||||
$$
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #BA2121; font-style: italic">"""The sigmoid function (or the logistic curve) is a</span>
|
||||
<span style="color: #BA2121; font-style: italic">function that takes any real number, z, and outputs a number (0,1).</span>
|
||||
<span style="color: #BA2121; font-style: italic">It is useful in neural networks for assigning weights on a relative scale.</span>
|
||||
<span style="color: #BA2121; font-style: italic">The value z is the weighted sum of parameters involved in the learning algorithm."""</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">math</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">mt</span>
|
||||
|
||||
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">.1</span>)
|
||||
sigma_fn <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>vectorize(<span style="color: #008000; font-weight: bold">lambda</span> z: <span style="color: #666666">1/</span>(<span style="color: #666666">1+</span>numpy<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>z)))
|
||||
sigma <span style="color: #666666">=</span> sigma_fn(z)
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
||||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
||||
ax<span style="color: #666666">.</span>plot(z, sigma)
|
||||
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-0.1</span>, <span style="color: #666666">1.1</span>])
|
||||
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-5</span>,<span style="color: #666666">5</span>])
|
||||
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
||||
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
||||
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'sigmoid function'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
|
||||
<span style="color: #BA2121; font-style: italic">"""Step Function"""</span>
|
||||
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-5</span>, <span style="color: #666666">5</span>, <span style="color: #666666">.02</span>)
|
||||
step_fn <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>vectorize(<span style="color: #008000; font-weight: bold">lambda</span> z: <span style="color: #666666">1.0</span> <span style="color: #008000; font-weight: bold">if</span> z <span style="color: #666666">>=</span> <span style="color: #666666">0.0</span> <span style="color: #008000; font-weight: bold">else</span> <span style="color: #666666">0.0</span>)
|
||||
step <span style="color: #666666">=</span> step_fn(z)
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
||||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
||||
ax<span style="color: #666666">.</span>plot(z, step)
|
||||
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-0.5</span>, <span style="color: #666666">1.5</span>])
|
||||
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-5</span>,<span style="color: #666666">5</span>])
|
||||
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
||||
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
||||
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'step function'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
|
||||
<span style="color: #BA2121; font-style: italic">"""tanh Function"""</span>
|
||||
z <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>arange(<span style="color: #666666">-2*</span>mt<span style="color: #666666">.</span>pi, <span style="color: #666666">2*</span>mt<span style="color: #666666">.</span>pi, <span style="color: #666666">0.1</span>)
|
||||
t <span style="color: #666666">=</span> numpy<span style="color: #666666">.</span>tanh(z)
|
||||
|
||||
fig <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>figure()
|
||||
ax <span style="color: #666666">=</span> fig<span style="color: #666666">.</span>add_subplot(<span style="color: #666666">111</span>)
|
||||
ax<span style="color: #666666">.</span>plot(z, t)
|
||||
ax<span style="color: #666666">.</span>set_ylim([<span style="color: #666666">-1.0</span>, <span style="color: #666666">1.0</span>])
|
||||
ax<span style="color: #666666">.</span>set_xlim([<span style="color: #666666">-2*</span>mt<span style="color: #666666">.</span>pi,<span style="color: #666666">2*</span>mt<span style="color: #666666">.</span>pi])
|
||||
ax<span style="color: #666666">.</span>grid(<span style="color: #008000; font-weight: bold">True</span>)
|
||||
ax<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">'z'</span>)
|
||||
ax<span style="color: #666666">.</span>set_title(<span style="color: #BA2121">'tanh function'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -455,7 +503,7 @@ $$
|
||||
<li><a href="._week38-bs041.html">42</a></li>
|
||||
<li><a href="._week38-bs042.html">43</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs034.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,24 +414,25 @@ MathJax.Hub.Config({
|
||||
<a name="part0034"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="the-cost-function-rewritten" class="anchor">The cost function rewritten </h2>
|
||||
<h2 id="two-parameters" class="anchor">Two parameters </h2>
|
||||
|
||||
<p>
|
||||
Reordering the logarithms, we can rewrite the <b>cost/loss</b> function as
|
||||
We assume now that we have two classes with \( y_i \) either \( 0 \) or \( 1 \). Furthermore we assume also that we have only two parameters \( \beta \) in our fitting of the Sigmoid function, that is we define probabilities
|
||||
$$
|
||||
\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
|
||||
\begin{align*}
|
||||
p(y_i=1|x_i,\hat{\beta}) &= \frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}},\nonumber\\
|
||||
p(y_i=0|x_i,\hat{\beta}) &= 1 - p(y_i=1|x_i,\hat{\beta}),
|
||||
\end{align*}
|
||||
$$
|
||||
|
||||
where \( \hat{\beta} \) are the weights we wish to extract from data, in our case \( \beta_0 \) and \( \beta_1 \).
|
||||
|
||||
<p>
|
||||
The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \).
|
||||
Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that
|
||||
Note that we used
|
||||
$$
|
||||
\mathcal{C}(\hat{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
|
||||
p(y_i=0\vert x_i, \hat{\beta}) = 1-p(y_i=1\vert x_i, \hat{\beta}).
|
||||
$$
|
||||
|
||||
This equation is known in statistics as the <b>cross entropy</b>. Finally, we note that just as in linear regression,
|
||||
in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -453,7 +459,7 @@ in practice we often supplement the cross-entropy with additional regularization
|
||||
<li><a href="._week38-bs042.html">43</a></li>
|
||||
<li><a href="._week38-bs043.html">44</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs035.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -407,25 +412,26 @@ MathJax.Hub.Config({
|
||||
<p> </p><p> </p><p> </p> <!-- add vertical space -->
|
||||
|
||||
<a name="part0035"></a>
|
||||
<!-- !split -->
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="minimizing-the-cross-entropy" class="anchor">Minimizing the cross entropy </h2>
|
||||
<h2 id="maximum-likelihood" class="anchor">Maximum likelihood </h2>
|
||||
|
||||
<p>
|
||||
The cross entropy is a convex function of the weights \( \hat{\beta} \) and,
|
||||
therefore, any local minimizer is a global minimizer.
|
||||
|
||||
<p>
|
||||
Minimizing this
|
||||
cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain
|
||||
|
||||
In order to define the total likelihood for all possible outcomes from a
|
||||
dataset \( \mathcal{D}=\{(y_i,x_i)\} \), with the binary labels
|
||||
\( y_i\in\{0,1\} \) and where the data points are drawn independently, we use the so-called <a href="https://en.wikipedia.org/wiki/Maximum_likelihood_estimation" target="_self">Maximum Likelihood Estimation</a> (MLE) principle.
|
||||
We aim thus at maximizing
|
||||
the probability of seeing the observed data. We can then approximate the
|
||||
likelihood in terms of the product of the individual probabilities of a specific outcome \( y_i \), that is
|
||||
$$
|
||||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right),
|
||||
\begin{align*}
|
||||
P(\mathcal{D}|\hat{\beta})& = \prod_{i=1}^n \left[p(y_i=1|x_i,\hat{\beta})\right]^{y_i}\left[1-p(y_i=1|x_i,\hat{\beta}))\right]^{1-y_i}\nonumber \\
|
||||
\end{align*}
|
||||
$$
|
||||
|
||||
and
|
||||
from which we obtain the log-likelihood and our <b>cost/loss</b> function
|
||||
$$
|
||||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right).
|
||||
\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left( y_i\log{p(y_i=1|x_i,\hat{\beta})} + (1-y_i)\log\left[1-p(y_i=1|x_i,\hat{\beta}))\right]\right).
|
||||
$$
|
||||
|
||||
<p>
|
||||
@@ -454,7 +460,7 @@ $$
|
||||
<li><a href="._week38-bs043.html">44</a></li>
|
||||
<li><a href="._week38-bs044.html">45</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs036.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,25 +414,23 @@ MathJax.Hub.Config({
|
||||
<a name="part0036"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="a-more-compact-expression" class="anchor">A more compact expression </h2>
|
||||
<h2 id="the-cost-function-rewritten" class="anchor">The cost function rewritten </h2>
|
||||
|
||||
<p>
|
||||
Let us now define a vector \( \hat{y} \) with \( n \) elements \( y_i \), an
|
||||
\( n\times p \) matrix \( \hat{X} \) which contains the \( x_i \) values and a
|
||||
vector \( \hat{p} \) of fitted probabilities \( p(y_i\vert x_i,\hat{\beta}) \). We can rewrite in a more compact form the first
|
||||
derivative of cost function as
|
||||
|
||||
Reordering the logarithms, we can rewrite the <b>cost/loss</b> function as
|
||||
$$
|
||||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right).
|
||||
\mathcal{C}(\hat{\beta}) = \sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
|
||||
$$
|
||||
|
||||
<p>
|
||||
If we in addition define a diagonal matrix \( \hat{W} \) with elements
|
||||
\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as
|
||||
The maximum likelihood estimator is defined as the set of parameters that maximize the log-likelihood where we maximize with respect to \( \beta \).
|
||||
Since the cost (error) function is just the negative log-likelihood, for logistic regression we have that
|
||||
$$
|
||||
\mathcal{C}(\hat{\beta})=-\sum_{i=1}^n \left(y_i(\beta_0+\beta_1x_i) -\log{(1+\exp{(\beta_0+\beta_1x_i)})}\right).
|
||||
$$
|
||||
|
||||
$$
|
||||
\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}.
|
||||
$$
|
||||
This equation is known in statistics as the <b>cross entropy</b>. Finally, we note that just as in linear regression,
|
||||
in practice we often supplement the cross-entropy with additional regularization terms, usually \( L_1 \) and \( L_2 \) regularization as we did for Ridge and Lasso regression.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -455,7 +458,7 @@ $$
|
||||
<li><a href="._week38-bs044.html">45</a></li>
|
||||
<li><a href="._week38-bs045.html">46</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs037.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,17 +414,23 @@ MathJax.Hub.Config({
|
||||
<a name="part0037"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="extending-to-more-predictors" class="anchor">Extending to more predictors </h2>
|
||||
<h2 id="minimizing-the-cross-entropy" class="anchor">Minimizing the cross entropy </h2>
|
||||
|
||||
<p>
|
||||
Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors
|
||||
The cross entropy is a convex function of the weights \( \hat{\beta} \) and,
|
||||
therefore, any local minimizer is a global minimizer.
|
||||
|
||||
<p>
|
||||
Minimizing this
|
||||
cost function with respect to the two parameters \( \beta_0 \) and \( \beta_1 \) we obtain
|
||||
|
||||
$$
|
||||
\log{ \frac{p(\hat{\beta}\hat{x})}{1-p(\hat{\beta}\hat{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p.
|
||||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_0} = -\sum_{i=1}^n \left(y_i -\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right),
|
||||
$$
|
||||
|
||||
Here we defined \( \hat{x}=[1,x_1,x_2,\dots,x_p] \) and \( \hat{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to
|
||||
and
|
||||
$$
|
||||
p(\hat{\beta}\hat{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}.
|
||||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \beta_1} = -\sum_{i=1}^n \left(y_ix_i -x_i\frac{\exp{(\beta_0+\beta_1x_i)}}{1+\exp{(\beta_0+\beta_1x_i)}}\right).
|
||||
$$
|
||||
|
||||
<p>
|
||||
@@ -448,7 +459,7 @@ $$
|
||||
<li><a href="._week38-bs045.html">46</a></li>
|
||||
<li><a href="._week38-bs046.html">47</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs038.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,30 +414,25 @@ MathJax.Hub.Config({
|
||||
<a name="part0038"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="including-more-classes" class="anchor">Including more classes </h2>
|
||||
<h2 id="a-more-compact-expression" class="anchor">A more compact expression </h2>
|
||||
|
||||
<p>
|
||||
Till now we have mainly focused on two classes, the so-called binary
|
||||
system. Suppose we wish to extend to \( K \) classes. Let us for the sake
|
||||
of simplicity assume we have only two predictors. We have then following model
|
||||
Let us now define a vector \( \hat{y} \) with \( n \) elements \( y_i \), an
|
||||
\( n\times p \) matrix \( \hat{X} \) which contains the \( x_i \) values and a
|
||||
vector \( \hat{p} \) of fitted probabilities \( p(y_i\vert x_i,\hat{\beta}) \). We can rewrite in a more compact form the first
|
||||
derivative of cost function as
|
||||
|
||||
$$
|
||||
\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1,
|
||||
$$
|
||||
|
||||
and
|
||||
$$
|
||||
\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1,
|
||||
$$
|
||||
|
||||
and so on till the class \( C=K-1 \) class
|
||||
$$
|
||||
\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1,
|
||||
\frac{\partial \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}} = -\hat{X}^T\left(\hat{y}-\hat{p}\right).
|
||||
$$
|
||||
|
||||
<p>
|
||||
and the model is specified in term of \( K-1 \) so-called log-odds or
|
||||
<b>logit</b> transformations.
|
||||
If we in addition define a diagonal matrix \( \hat{W} \) with elements
|
||||
\( p(y_i\vert x_i,\hat{\beta})(1-p(y_i\vert x_i,\hat{\beta}) \), we can obtain a compact expression of the second derivative as
|
||||
|
||||
$$
|
||||
\frac{\partial^2 \mathcal{C}(\hat{\beta})}{\partial \hat{\beta}\partial \hat{\beta}^T} = \hat{X}^T\hat{W}\hat{X}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -460,7 +460,7 @@ and the model is specified in term of \( K-1 \) so-called log-odds or
|
||||
<li><a href="._week38-bs046.html">47</a></li>
|
||||
<li><a href="._week38-bs047.html">48</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs039.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,43 +414,19 @@ MathJax.Hub.Config({
|
||||
<a name="part0039"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="more-classes" class="anchor">More classes </h2>
|
||||
<h2 id="extending-to-more-predictors" class="anchor">Extending to more predictors </h2>
|
||||
|
||||
<p>
|
||||
In our discussion of neural networks we will encounter the above again
|
||||
in terms of a slightly modified function, the so-called <b>Softmax</b> function.
|
||||
|
||||
<p>
|
||||
The softmax function is used in various multiclass classification
|
||||
methods, such as multinomial logistic regression (also known as
|
||||
softmax regression), multiclass linear discriminant analysis, naive
|
||||
Bayes classifiers, and artificial neural networks. Specifically, in
|
||||
multinomial logistic regression and linear discriminant analysis, the
|
||||
input to the function is the result of \( K \) distinct linear functions,
|
||||
and the predicted probability for the \( k \)-th class given a sample
|
||||
vector \( \hat{x} \) and a weighting vector \( \hat{\beta} \) is (with two
|
||||
predictors):
|
||||
|
||||
Within a binary classification problem, we can easily expand our model to include multiple predictors. Our ratio between likelihoods is then with \( p \) predictors
|
||||
$$
|
||||
p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}.
|
||||
\log{ \frac{p(\hat{\beta}\hat{x})}{1-p(\hat{\beta}\hat{x})}} = \beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p.
|
||||
$$
|
||||
|
||||
It is easy to extend to more predictors. The final class is
|
||||
Here we defined \( \hat{x}=[1,x_1,x_2,\dots,x_p] \) and \( \hat{\beta}=[\beta_0, \beta_1, \dots, \beta_p] \) leading to
|
||||
$$
|
||||
p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}},
|
||||
p(\hat{\beta}\hat{x})=\frac{ \exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}{1+\exp{(\beta_0+\beta_1x_1+\beta_2x_2+\dots+\beta_px_p)}}.
|
||||
$$
|
||||
|
||||
<p>
|
||||
and they sum to one. Our earlier discussions were all specialized to
|
||||
the case with two classes only. It is easy to see from the above that
|
||||
what we derived earlier is compatible with these equations.
|
||||
|
||||
<p>
|
||||
To find the optimal parameters we would typically use a gradient
|
||||
descent method. Newton's method and gradient descent methods are
|
||||
discussed in the material on <a href="https://compphysics.github.io/MachineLearning/doc/pub/Splines/html/Splines-bs.html" target="_self">optimization
|
||||
methods</a>.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -472,7 +453,7 @@ methods</a>.
|
||||
<li><a href="._week38-bs047.html">48</a></li>
|
||||
<li><a href="._week38-bs048.html">49</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs040.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,7 +414,30 @@ MathJax.Hub.Config({
|
||||
<a name="part0040"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="friday-september-24" class="anchor">Friday September 24 </h2>
|
||||
<h2 id="including-more-classes" class="anchor">Including more classes </h2>
|
||||
|
||||
<p>
|
||||
Till now we have mainly focused on two classes, the so-called binary
|
||||
system. Suppose we wish to extend to \( K \) classes. Let us for the sake
|
||||
of simplicity assume we have only two predictors. We have then following model
|
||||
|
||||
$$
|
||||
\log{\frac{p(C=1\vert x)}{p(K\vert x)}} = \beta_{10}+\beta_{11}x_1,
|
||||
$$
|
||||
|
||||
and
|
||||
$$
|
||||
\log{\frac{p(C=2\vert x)}{p(K\vert x)}} = \beta_{20}+\beta_{21}x_1,
|
||||
$$
|
||||
|
||||
and so on till the class \( C=K-1 \) class
|
||||
$$
|
||||
\log{\frac{p(C=K-1\vert x)}{p(K\vert x)}} = \beta_{(K-1)0}+\beta_{(K-1)1}x_1,
|
||||
$$
|
||||
|
||||
<p>
|
||||
and the model is specified in term of \( K-1 \) so-called log-odds or
|
||||
<b>logit</b> transformations.
|
||||
|
||||
<p>
|
||||
<p>
|
||||
@@ -437,7 +465,7 @@ MathJax.Hub.Config({
|
||||
<li><a href="._week38-bs048.html">49</a></li>
|
||||
<li><a href="._week38-bs049.html">50</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs041.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,42 +414,43 @@ MathJax.Hub.Config({
|
||||
<a name="part0041"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="wisconsin-cancer-data" class="anchor">Wisconsin Cancer Data </h2>
|
||||
<h2 id="more-classes" class="anchor">More classes </h2>
|
||||
|
||||
<p>
|
||||
We show here how we can use a simple regression case on the breast
|
||||
cancer data using Logistic regression as our algorithm for
|
||||
classification.
|
||||
In our discussion of neural networks we will encounter the above again
|
||||
in terms of a slightly modified function, the so-called <b>Softmax</b> function.
|
||||
|
||||
<p>
|
||||
The softmax function is used in various multiclass classification
|
||||
methods, such as multinomial logistic regression (also known as
|
||||
softmax regression), multiclass linear discriminant analysis, naive
|
||||
Bayes classifiers, and artificial neural networks. Specifically, in
|
||||
multinomial logistic regression and linear discriminant analysis, the
|
||||
input to the function is the result of \( K \) distinct linear functions,
|
||||
and the predicted probability for the \( k \)-th class given a sample
|
||||
vector \( \hat{x} \) and a weighting vector \( \hat{\beta} \) is (with two
|
||||
predictors):
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.datasets</span> <span style="color: #008000; font-weight: bold">import</span> load_breast_cancer
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LogisticRegression
|
||||
$$
|
||||
p(C=k\vert \mathbf {x} )=\frac{\exp{(\beta_{k0}+\beta_{k1}x_1)}}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}}.
|
||||
$$
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Load the data</span>
|
||||
cancer <span style="color: #666666">=</span> load_breast_cancer()
|
||||
It is easy to extend to more predictors. The final class is
|
||||
$$
|
||||
p(C=K\vert \mathbf {x} )=\frac{1}{1+\sum_{l=1}^{K-1}\exp{(\beta_{l0}+\beta_{l1}x_1)}},
|
||||
$$
|
||||
|
||||
<p>
|
||||
and they sum to one. Our earlier discussions were all specialized to
|
||||
the case with two classes only. It is easy to see from the above that
|
||||
what we derived earlier is compatible with these equations.
|
||||
|
||||
<p>
|
||||
To find the optimal parameters we would typically use a gradient
|
||||
descent method. Newton's method and gradient descent methods are
|
||||
discussed in the material on <a href="https://compphysics.github.io/MachineLearning/doc/pub/Splines/html/Splines-bs.html" target="_self">optimization
|
||||
methods</a>.
|
||||
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(cancer<span style="color: #666666">.</span>data,cancer<span style="color: #666666">.</span>target,random_state<span style="color: #666666">=0</span>)
|
||||
<span style="color: #008000">print</span>(X_train<span style="color: #666666">.</span>shape)
|
||||
<span style="color: #008000">print</span>(X_test<span style="color: #666666">.</span>shape)
|
||||
<span style="color: #408080; font-style: italic"># Logistic Regression</span>
|
||||
logreg <span style="color: #666666">=</span> LogisticRegression(solver<span style="color: #666666">=</span><span style="color: #BA2121">'lbfgs'</span>)
|
||||
logreg<span style="color: #666666">.</span>fit(X_train, y_train)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test set accuracy with Logistic Regression: </span><span style="color: #BB6688; font-weight: bold">{:.2f}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(logreg<span style="color: #666666">.</span>score(X_test,y_test)))
|
||||
<span style="color: #408080; font-style: italic">#now scale the data</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
scaler <span style="color: #666666">=</span> StandardScaler()
|
||||
scaler<span style="color: #666666">.</span>fit(X_train)
|
||||
X_train_scaled <span style="color: #666666">=</span> scaler<span style="color: #666666">.</span>transform(X_train)
|
||||
X_test_scaled <span style="color: #666666">=</span> scaler<span style="color: #666666">.</span>transform(X_test)
|
||||
<span style="color: #408080; font-style: italic"># Logistic Regression</span>
|
||||
logreg<span style="color: #666666">.</span>fit(X_train_scaled, y_train)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test set accuracy Logistic Regression with scaled data: </span><span style="color: #BB6688; font-weight: bold">{:.2f}</span><span style="color: #BA2121">"</span><span style="color: #666666">.</span>format(logreg<span style="color: #666666">.</span>score(X_test_scaled,y_test)))
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -471,7 +477,7 @@ logreg<span style="color: #666666">.</span>fit(X_train_scaled, y_train)
|
||||
<li><a href="._week38-bs049.html">50</a></li>
|
||||
<li><a href="._week38-bs050.html">51</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs042.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -409,49 +414,8 @@ MathJax.Hub.Config({
|
||||
<a name="part0042"></a>
|
||||
<!-- !split -->
|
||||
|
||||
<h2 id="using-the-correlation-matrix" class="anchor">Using the correlation matrix </h2>
|
||||
<h2 id="friday-september-24" class="anchor">Friday September 24 </h2>
|
||||
|
||||
<p>
|
||||
In addition to the above scores, we could also study the covariance (and the correlation matrix).
|
||||
We use <b>Pandas</b> to compute the correlation matrix.
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">matplotlib.pyplot</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">plt</span>
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.datasets</span> <span style="color: #008000; font-weight: bold">import</span> load_breast_cancer
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.linear_model</span> <span style="color: #008000; font-weight: bold">import</span> LogisticRegression
|
||||
cancer <span style="color: #666666">=</span> load_breast_cancer()
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">pandas</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">pd</span>
|
||||
<span style="color: #408080; font-style: italic"># Making a data frame</span>
|
||||
cancerpd <span style="color: #666666">=</span> pd<span style="color: #666666">.</span>DataFrame(cancer<span style="color: #666666">.</span>data, columns<span style="color: #666666">=</span>cancer<span style="color: #666666">.</span>feature_names)
|
||||
|
||||
fig, axes <span style="color: #666666">=</span> plt<span style="color: #666666">.</span>subplots(<span style="color: #666666">15</span>,<span style="color: #666666">2</span>,figsize<span style="color: #666666">=</span>(<span style="color: #666666">10</span>,<span style="color: #666666">20</span>))
|
||||
malignant <span style="color: #666666">=</span> cancer<span style="color: #666666">.</span>data[cancer<span style="color: #666666">.</span>target <span style="color: #666666">==</span> <span style="color: #666666">0</span>]
|
||||
benign <span style="color: #666666">=</span> cancer<span style="color: #666666">.</span>data[cancer<span style="color: #666666">.</span>target <span style="color: #666666">==</span> <span style="color: #666666">1</span>]
|
||||
ax <span style="color: #666666">=</span> axes<span style="color: #666666">.</span>ravel()
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">30</span>):
|
||||
_, bins <span style="color: #666666">=</span> np<span style="color: #666666">.</span>histogram(cancer<span style="color: #666666">.</span>data[:,i], bins <span style="color: #666666">=50</span>)
|
||||
ax[i]<span style="color: #666666">.</span>hist(malignant[:,i], bins <span style="color: #666666">=</span> bins, alpha <span style="color: #666666">=</span> <span style="color: #666666">0.5</span>)
|
||||
ax[i]<span style="color: #666666">.</span>hist(benign[:,i], bins <span style="color: #666666">=</span> bins, alpha <span style="color: #666666">=</span> <span style="color: #666666">0.5</span>)
|
||||
ax[i]<span style="color: #666666">.</span>set_title(cancer<span style="color: #666666">.</span>feature_names[i])
|
||||
ax[i]<span style="color: #666666">.</span>set_yticks(())
|
||||
ax[<span style="color: #666666">0</span>]<span style="color: #666666">.</span>set_xlabel(<span style="color: #BA2121">"Feature magnitude"</span>)
|
||||
ax[<span style="color: #666666">0</span>]<span style="color: #666666">.</span>set_ylabel(<span style="color: #BA2121">"Frequency"</span>)
|
||||
ax[<span style="color: #666666">0</span>]<span style="color: #666666">.</span>legend([<span style="color: #BA2121">"Malignant"</span>, <span style="color: #BA2121">"Benign"</span>], loc <span style="color: #666666">=</span><span style="color: #BA2121">"best"</span>)
|
||||
fig<span style="color: #666666">.</span>tight_layout()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
|
||||
<span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">seaborn</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">sns</span>
|
||||
correlation_matrix <span style="color: #666666">=</span> cancerpd<span style="color: #666666">.</span>corr()<span style="color: #666666">.</span>round(<span style="color: #666666">1</span>)
|
||||
<span style="color: #408080; font-style: italic"># use the heatmap function from seaborn to plot the correlation matrix</span>
|
||||
<span style="color: #408080; font-style: italic"># annot = True to print the values inside the square</span>
|
||||
plt<span style="color: #666666">.</span>figure(figsize<span style="color: #666666">=</span>(<span style="color: #666666">15</span>,<span style="color: #666666">8</span>))
|
||||
sns<span style="color: #666666">.</span>heatmap(data<span style="color: #666666">=</span>correlation_matrix, annot<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">True</span>)
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
<p>
|
||||
<!-- navigation buttons at the bottom of the page -->
|
||||
@@ -478,7 +442,7 @@ plt<span style="color: #666666">.</span>show()
|
||||
<li><a href="._week38-bs050.html">51</a></li>
|
||||
<li><a href="._week38-bs051.html">52</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs043.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -74,10 +74,13 @@ Automatically generated HTML file from DocOnce source
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -319,81 +322,83 @@ MathJax.Hub.Config({
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs009.html#more-thinking" style="font-size: 80%;">More thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs010.html#still-thinking" style="font-size: 80%;">Still thinking</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs011.html#linear-regression-code-intercept-handling-first" style="font-size: 80%;">Linear Regression code, Intercept handling first</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-mean-mathematically" style="font-size: 80%;">What does centering mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs067.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs073.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs076.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs077.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs080.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs012.html#what-does-centering-subtracting-the-mean-values-mean-mathematically" style="font-size: 80%;">What does centering (subtracting the mean values) mean mathematically?</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs013.html#code-examples" style="font-size: 80%;">Code Examples</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs014.html#taking-out-the-mean" style="font-size: 80%;">Taking out the mean</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs015.html#more-complicated-example-the-ising-model" style="font-size: 80%;">More complicated Example: The Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs016.html#reformulating-the-problem-to-suit-regression" style="font-size: 80%;">Reformulating the problem to suit regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs017.html#linear-regression" style="font-size: 80%;">Linear regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs018.html#singular-value-decomposition" style="font-size: 80%;">Singular Value decomposition</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs019.html#the-one-dimensional-ising-model" style="font-size: 80%;">The one-dimensional Ising model</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs020.html#ridge-regression" style="font-size: 80%;">Ridge regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs021.html#lasso-regression" style="font-size: 80%;">LASSO regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs022.html#performance-as-function-of-the-regularization-parameter" style="font-size: 80%;">Performance as function of the regularization parameter</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs023.html#finding-the-optimal-value-of-lambda" style="font-size: 80%;">Finding the optimal value of \( \lambda \)</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs024.html#logistic-regression" style="font-size: 80%;">Logistic Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs025.html#classification-problems" style="font-size: 80%;">Classification problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs026.html#optimization-and-deep-learning" style="font-size: 80%;">Optimization and Deep learning</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs027.html#basics" style="font-size: 80%;">Basics</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs028.html#linear-classifier" style="font-size: 80%;">Linear classifier</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs029.html#some-selected-properties" style="font-size: 80%;">Some selected properties</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs030.html#simple-example" style="font-size: 80%;">Simple example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs031.html#plotting-the-mean-value-for-each-group" style="font-size: 80%;">Plotting the mean value for each group</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs032.html#the-logistic-function" style="font-size: 80%;">The logistic function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs033.html#examples-of-likelihood-functions-used-in-logistic-regression-and-nueral-networks" style="font-size: 80%;">Examples of likelihood functions used in logistic regression and nueral networks</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs034.html#two-parameters" style="font-size: 80%;">Two parameters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs035.html#maximum-likelihood" style="font-size: 80%;">Maximum likelihood</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs036.html#the-cost-function-rewritten" style="font-size: 80%;">The cost function rewritten</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs037.html#minimizing-the-cross-entropy" style="font-size: 80%;">Minimizing the cross entropy</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs038.html#a-more-compact-expression" style="font-size: 80%;">A more compact expression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs039.html#extending-to-more-predictors" style="font-size: 80%;">Extending to more predictors</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs040.html#including-more-classes" style="font-size: 80%;">Including more classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs041.html#more-classes" style="font-size: 80%;">More classes</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs042.html#friday-september-24" style="font-size: 80%;">Friday September 24</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs043.html#wisconsin-cancer-data" style="font-size: 80%;">Wisconsin Cancer Data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs044.html#using-the-correlation-matrix" style="font-size: 80%;">Using the correlation matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs045.html#discussing-the-correlation-data" style="font-size: 80%;">Discussing the correlation data</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs046.html#other-measures-in-classification-studies-cancer-data-again" style="font-size: 80%;">Other measures in classification studies: Cancer Data again</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs047.html#optimization-the-central-part-of-any-machine-learning-algortithm" style="font-size: 80%;">Optimization, the central part of any Machine Learning algortithm</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs048.html#revisiting-our-logistic-regression-case" style="font-size: 80%;">Revisiting our Logistic Regression case</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs049.html#the-equations-to-solve" style="font-size: 80%;">The equations to solve</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs050.html#solving-using-newton-raphson-s-method" style="font-size: 80%;">Solving using Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs051.html#brief-reminder-on-newton-raphson-s-method" style="font-size: 80%;">Brief reminder on Newton-Raphson's method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs052.html#the-equations" style="font-size: 80%;">The equations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs053.html#simple-geometric-interpretation" style="font-size: 80%;">Simple geometric interpretation</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs054.html#extending-to-more-than-one-variable" style="font-size: 80%;">Extending to more than one variable</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs055.html#steepest-descent" style="font-size: 80%;">Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs056.html#more-on-steepest-descent" style="font-size: 80%;">More on Steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs057.html#the-ideal" style="font-size: 80%;">The ideal</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs058.html#the-sensitiveness-of-the-gradient-descent" style="font-size: 80%;">The sensitiveness of the gradient descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs059.html#convex-functions" style="font-size: 80%;">Convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs060.html#convex-function" style="font-size: 80%;">Convex function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs061.html#conditions-on-convex-functions" style="font-size: 80%;">Conditions on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs062.html#more-on-convex-functions" style="font-size: 80%;">More on convex functions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs063.html#some-simple-problems" style="font-size: 80%;">Some simple problems</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs064.html#friday-september-25" style="font-size: 80%;">Friday September 25</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs065.html#standard-steepest-descent" style="font-size: 80%;">Standard steepest descent</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs066.html#gradient-method" style="font-size: 80%;">Gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs068.html#steepest-descent-method" style="font-size: 80%;">Steepest descent method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs069.html#final-expressions" style="font-size: 80%;">Final expressions</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs070.html#steepest-descent-example" style="font-size: 80%;">Steepest descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs075.html#conjugate-gradient-method-and-iterations" style="font-size: 80%;">Conjugate gradient method and iterations</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs078.html#conjugate-gradient-method" style="font-size: 80%;">Conjugate gradient method</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs079.html#revisiting-some-of-our-first-linear-regression-encounters" style="font-size: 80%;">Revisiting some of our first Linear Regression Encounters</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient descent example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs081.html#the-derivative-of-the-cost-loss-function" style="font-size: 80%;">The derivative of the cost/loss function</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs082.html#the-hessian-matrix" style="font-size: 80%;">The Hessian matrix</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs083.html#simple-program" style="font-size: 80%;">Simple program</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs084.html#gradient-descent-example" style="font-size: 80%;">Gradient Descent Example</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs085.html#and-a-corresponding-example-using-_scikit-learn_" style="font-size: 80%;">And a corresponding example using <b>scikit-learn</b></a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs086.html#gradient-descent-and-ridge" style="font-size: 80%;">Gradient descent and Ridge</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs087.html#program-example-for-gradient-descent-with-ridge-regression" style="font-size: 80%;">Program example for gradient descent with Ridge Regression</a></li>
|
||||
<!-- navigation toc: --> <li><a href="._week38-bs088.html#using-gradient-descent-methods-limitations" style="font-size: 80%;">Using gradient descent methods, limitations</a></li>
|
||||
|
||||
</ul>
|
||||
</li>
|
||||
@@ -452,7 +457,7 @@ MathJax.Hub.Config({
|
||||
<li><a href="._week38-bs008.html">9</a></li>
|
||||
<li><a href="._week38-bs009.html">10</a></li>
|
||||
<li><a href="">...</a></li>
|
||||
<li><a href="._week38-bs086.html">87</a></li>
|
||||
<li><a href="._week38-bs088.html">89</a></li>
|
||||
<li><a href="._week38-bs001.html">»</a></li>
|
||||
</ul>
|
||||
<!-- ------------------- end of main content --------------- -->
|
||||
|
||||
@@ -458,14 +458,16 @@ $$\beta_0$$
|
||||
If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation.
|
||||
standard deviation. Most machine learning libraries do this as a deafult. This means that if you compare your code with the results from a given library,
|
||||
the results may differ. Tracing back the differences may often lead to an increased confusion.
|
||||
|
||||
<p>
|
||||
The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_blank">Standadscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them.
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
|
||||
<p>
|
||||
If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
@@ -515,7 +517,7 @@ y_pred = y_pred + y_train_mean
|
||||
<h2 id="linear-regression-code-intercept-handling-first">Linear Regression code, Intercept handling first </h2>
|
||||
|
||||
<p>
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>)
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
|
||||
<p>
|
||||
|
||||
@@ -595,42 +597,35 @@ plt.show()
|
||||
|
||||
|
||||
<section>
|
||||
<h2 id="what-does-centering-mean-mathematically">What does centering mean mathematically? </h2>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
<h2 id="what-does-centering-subtracting-the-mean-values-mean-mathematically">What does centering (subtracting the mean values) mean mathematically? </h2>
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
Let us try to understand what this may imply mathematically when we subtract the mean values, also known as <em>zero centering</em>. To catch many birds with just one stone, we will focus on Ridge regression.
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is
|
||||
<p> <br>
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2 + \lambda \sum_{j=1}^{p-1} \beta_i^2.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
Note that the intercept term $\beta_0$is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
for all \( j \), so let us start with \( \beta_0 \). This means that we have
|
||||
|
||||
<p> <br>
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
<p> <br>
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
\frac{\partial C}{\partial \beta_0} = -2\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right),
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
@@ -638,36 +633,30 @@ $$
|
||||
which gives
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
If we assume that every column of \( \boldsymbol{X} \) is centered, which we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span>X = X - np.mean(X,axis=<span style="color: #B452CD">0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
the sum \( \sum_{i=0}^{n-1} X_{ij} \)
|
||||
can be rewritten as
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
\sum_{i=0}^{n-1} \left(X_{ij} - \frac{1}{n}\sum_{i=0}^{n-1} X_{ij}) = \sum_{i=0}^{n-1} X_{ij} - \sum_{i=0}^{n-1} \frac{1}{n} \sum_{i=0}^{n-1}X_{ij},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
resulting in
|
||||
<p> <br>
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
\sum_{i=0}^{n-1} X_{ij} - n \frac{1}{n} \sum_{i=0}^{n-1}X_{ij} = 0.
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
@@ -675,19 +664,21 @@ $$
|
||||
Finally we have
|
||||
<p> <br>
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
n\beta_0 = \sum_{i=0}^{n-1} y_i - \sum_{j=1}^{p-1}\beta_j \sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
or
|
||||
<p> <br>
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1} y_i = \overline{\boldsymbol{y}},
|
||||
$$
|
||||
<p> <br>
|
||||
|
||||
the average value of \( \boldsymbol{y] \).
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - \overline{\boldsymbol{y}} \) in the cost function will give us (in vector-matrix disguise)
|
||||
<p> <br>
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
@@ -699,8 +690,16 @@ which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
</section>
|
||||
|
||||
|
||||
<section>
|
||||
<h2 id="code-examples">Code Examples </h2>
|
||||
|
||||
<p>
|
||||
Armed with this wisdom, we attempt first simply set the intercept eqault to <b>False</b> in our implementation of Ridge regression for a vanilla data set.
|
||||
|
||||
<p>
|
||||
|
||||
@@ -711,8 +710,6 @@ and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">R2</span>(y_data, y_model):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span> - np.sum((y_data - y_model) ** <span style="color: #B452CD">2</span>) / np.sum((y_data - np.mean(y_data)) ** <span style="color: #B452CD">2</span>)
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
@@ -728,42 +725,29 @@ y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B45
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree))
|
||||
We include explicitely the intercpt column
|
||||
X[:,<span style="color: #B452CD">0</span>] = <span style="color: #B452CD">1.0</span>
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> polydegree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>, Maxpolydegree):
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(polydegree):
|
||||
X[:,degree] = x**degree
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(Maxpolydegree):
|
||||
X[:,degree] = x**degree
|
||||
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
<span style="color: #228B22"># matrix inversion to find beta</span>
|
||||
OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
|
||||
<span style="color: #658b00">print</span>(OLSbeta)
|
||||
<span style="color: #228B22"># and then make the prediction</span>
|
||||
ytildeOLS = X_train @ OLSbeta
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Training MSE for OLS"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y_train,ytildeOLS))
|
||||
ypredictOLS = X_test @ OLSbeta
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Test MSE OLS"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y_test,ypredictOLS))
|
||||
|
||||
p = <span style="color: #658b00">len</span>(OLSbeta)
|
||||
p = Maxpolydegree
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">4</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSEOwnRidgeTrain = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
MSERidgeTrain = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">4</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
|
||||
<span style="color: #228B22"># include lasso using Scikit-Learn</span>
|
||||
<span style="color: #228B22"># Note: we include the intercept</span>
|
||||
<span style="color: #228B22"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge = linear_model.Ridge(lmb,fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
<span style="color: #228B22"># and then make the prediction</span>
|
||||
@@ -772,18 +756,14 @@ lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color
|
||||
ytildeRidge = RegRidge.predict(X_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, <span style="color: #CD5555">'b'</span>, label = <span style="color: #CD5555">'MSE Ridge train'</span>)
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'r'</span>, label = <span style="color: #CD5555">'MSE Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgeTrain, <span style="color: #CD5555">'y'</span>, label = <span style="color: #CD5555">'MSE Ridge train'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g'</span>, label = <span style="color: #CD5555">'MSE Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
@@ -792,6 +772,17 @@ plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
The problem however is that can easily lead to a larger mean-squared error!
|
||||
|
||||
<p>
|
||||
Let us see how we can change this code by zero centering.
|
||||
</section>
|
||||
|
||||
|
||||
<section>
|
||||
<h2 id="taking-out-the-mean">Taking out the mean </h2>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="font-size: 80%; line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
@@ -801,13 +792,9 @@ plt.show()
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">R2</span>(y_data, y_model):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span> - np.sum((y_data - y_model) ** <span style="color: #B452CD">2</span>) / np.sum((y_data - np.mean(y_data)) ** <span style="color: #B452CD">2</span>)
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">315</span>)
|
||||
@@ -816,15 +803,12 @@ n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">5</span>
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree-<span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,Maxpolydegree): <span style="color: #228B22">#No intercept column</span>
|
||||
X[:,degree-<span style="color: #B452CD">1</span>] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
@@ -834,11 +818,13 @@ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style=
|
||||
|
||||
<span style="color: #228B22">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
X_train_scaled = X_train - X_train_mean <span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
<span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) <span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled = y_train - y_scaler <span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
<span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
<span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
|
||||
p = Maxpolydegree-<span style="color: #B452CD">1</span>
|
||||
@@ -853,25 +839,20 @@ lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean<span style="color: #707a7c">@OwnRidgeBeta</span> <span style="color: #228B22">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #228B22">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_
|
||||
<span style="color: #228B22">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #228B22">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta) <span style="color: #228B22">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
@@ -881,14 +862,10 @@ lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'b--'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g--'</span>, label = <span style="color: #CD5555">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
|
||||
@@ -94,10 +94,13 @@ div { text-align: justify; text-justify: inter-word; }
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -616,14 +619,16 @@ Furthermore, in for example Ridge and Lasso regression, the solutions
|
||||
If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation.
|
||||
standard deviation. Most machine learning libraries do this as a deafult. This means that if you compare your code with the results from a given library,
|
||||
the results may differ. Tracing back the differences may often lead to an increased confusion.
|
||||
|
||||
<p>
|
||||
The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_blank">Standadscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them.
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
|
||||
<p>
|
||||
If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
@@ -672,7 +677,7 @@ y_pred = y_pred + y_train_mean
|
||||
<h2 id="linear-regression-code-intercept-handling-first">Linear Regression code, Intercept handling first </h2>
|
||||
|
||||
<p>
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>)
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
|
||||
<p>
|
||||
|
||||
@@ -751,81 +756,72 @@ plt.show()
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="what-does-centering-mean-mathematically">What does centering mean mathematically? </h2>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
<h2 id="what-does-centering-subtracting-the-mean-values-mean-mathematically">What does centering (subtracting the mean values) mean mathematically? </h2>
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
Let us try to understand what this may imply mathematically when we subtract the mean values, also known as <em>zero centering</em>. To catch many birds with just one stone, we will focus on Ridge regression.
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2 + \lambda \sum_{j=1}^{p-1} \beta_i^2.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
Note that the intercept term $\beta_0$is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
for all \( j \), so let us start with \( \beta_0 \). This means that we have
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
\frac{\partial C}{\partial \beta_0} = -2\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right),
|
||||
$$
|
||||
|
||||
<p>
|
||||
which gives
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
If we assume that every column of \( \boldsymbol{X} \) is centered, which we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span>X = X - np.mean(X,axis=<span style="color: #B452CD">0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
the sum \( \sum_{i=0}^{n-1} X_{ij} \)
|
||||
can be rewritten as
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
\sum_{i=0}^{n-1} \left(X_{ij} - \frac{1}{n}\sum_{i=0}^{n-1} X_{ij}) = \sum_{i=0}^{n-1} X_{ij} - \sum_{i=0}^{n-1} \frac{1}{n} \sum_{i=0}^{n-1}X_{ij},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
\sum_{i=0}^{n-1} X_{ij} - n \frac{1}{n} \sum_{i=0}^{n-1}X_{ij} = 0.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
n\beta_0 = \sum_{i=0}^{n-1} y_i - \sum_{j=1}^{p-1}\beta_j \sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
|
||||
or
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1} y_i = \overline{\boldsymbol{y}},
|
||||
$$
|
||||
|
||||
the average value of \( \boldsymbol{y] \).
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - \overline{\boldsymbol{y}} \) in the cost function will give us (in vector-matrix disguise)
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
@@ -835,9 +831,17 @@ which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="code-examples">Code Examples </h2>
|
||||
|
||||
<p>
|
||||
Armed with this wisdom, we attempt first simply set the intercept eqault to <b>False</b> in our implementation of Ridge regression for a vanilla data set.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
@@ -847,8 +851,6 @@ and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.model_selection</span> <span style="color: #8B008B; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">R2</span>(y_data, y_model):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span> - np.sum((y_data - y_model) ** <span style="color: #B452CD">2</span>) / np.sum((y_data - np.mean(y_data)) ** <span style="color: #B452CD">2</span>)
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
@@ -864,42 +866,29 @@ y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B45
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree))
|
||||
We include explicitely the intercpt column
|
||||
X[:,<span style="color: #B452CD">0</span>] = <span style="color: #B452CD">1.0</span>
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> polydegree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>, Maxpolydegree):
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(polydegree):
|
||||
X[:,degree] = x**degree
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(Maxpolydegree):
|
||||
X[:,degree] = x**degree
|
||||
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
<span style="color: #228B22"># matrix inversion to find beta</span>
|
||||
OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
|
||||
<span style="color: #658b00">print</span>(OLSbeta)
|
||||
<span style="color: #228B22"># and then make the prediction</span>
|
||||
ytildeOLS = X_train @ OLSbeta
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Training MSE for OLS"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y_train,ytildeOLS))
|
||||
ypredictOLS = X_test @ OLSbeta
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Test MSE OLS"</span>)
|
||||
<span style="color: #658b00">print</span>(MSE(y_test,ypredictOLS))
|
||||
|
||||
p = <span style="color: #658b00">len</span>(OLSbeta)
|
||||
p = Maxpolydegree
|
||||
I = np.eye(p,p)
|
||||
<span style="color: #228B22"># Decide which values of lambda to use</span>
|
||||
nlambdas = <span style="color: #B452CD">4</span>
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSEOwnRidgeTrain = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
MSERidgeTrain = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color: #B452CD">4</span>, nlambdas)
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> i <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
|
||||
<span style="color: #228B22"># include lasso using Scikit-Learn</span>
|
||||
<span style="color: #228B22"># Note: we include the intercept</span>
|
||||
<span style="color: #228B22"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge = linear_model.Ridge(lmb,fit_intercept=<span style="color: #8B008B; font-weight: bold">False</span>)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
<span style="color: #228B22"># and then make the prediction</span>
|
||||
@@ -908,18 +897,14 @@ lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color
|
||||
ytildeRidge = RegRidge.predict(X_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.coef_)
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, <span style="color: #CD5555">'b'</span>, label = <span style="color: #CD5555">'MSE Ridge train'</span>)
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'r'</span>, label = <span style="color: #CD5555">'MSE Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgeTrain, <span style="color: #CD5555">'y'</span>, label = <span style="color: #CD5555">'MSE Ridge train'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g'</span>, label = <span style="color: #CD5555">'MSE Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
@@ -928,6 +913,17 @@ plt.legend()
|
||||
plt.show()
|
||||
</pre></div>
|
||||
<p>
|
||||
The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
The problem however is that can easily lead to a larger mean-squared error!
|
||||
|
||||
<p>
|
||||
Let us see how we can change this code by zero centering.
|
||||
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="taking-out-the-mean">Taking out the mean </h2>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "perldoc" -->
|
||||
<div class="highlight" style="background: #eeeedd"><pre style="line-height: 125%;"><span></span><span style="color: #8B008B; font-weight: bold">import</span> <span style="color: #008b45; text-decoration: underline">numpy</span> <span style="color: #8B008B; font-weight: bold">as</span> <span style="color: #008b45; text-decoration: underline">np</span>
|
||||
@@ -937,13 +933,9 @@ plt.show()
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn</span> <span style="color: #8B008B; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #8B008B; font-weight: bold">from</span> <span style="color: #008b45; text-decoration: underline">sklearn.preprocessing</span> <span style="color: #8B008B; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">R2</span>(y_data, y_model):
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> <span style="color: #B452CD">1</span> - np.sum((y_data - y_model) ** <span style="color: #B452CD">2</span>) / np.sum((y_data - np.mean(y_data)) ** <span style="color: #B452CD">2</span>)
|
||||
<span style="color: #8B008B; font-weight: bold">def</span> <span style="color: #008b45">MSE</span>(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
<span style="color: #8B008B; font-weight: bold">return</span> np.sum((y_data-y_model)**<span style="color: #B452CD">2</span>)/n
|
||||
|
||||
|
||||
<span style="color: #228B22"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #228B22"># Useful for eventual debugging.</span>
|
||||
np.random.seed(<span style="color: #B452CD">315</span>)
|
||||
@@ -952,15 +944,12 @@ n = <span style="color: #B452CD">100</span>
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**<span style="color: #B452CD">2</span>) + <span style="color: #B452CD">1.5</span> * np.exp(-(x-<span style="color: #B452CD">2</span>)**<span style="color: #B452CD">2</span>)
|
||||
|
||||
Maxpolydegree = <span style="color: #B452CD">5</span>
|
||||
Maxpolydegree = <span style="color: #B452CD">20</span>
|
||||
X = np.zeros((n,Maxpolydegree-<span style="color: #B452CD">1</span>))
|
||||
|
||||
<span style="color: #8B008B; font-weight: bold">for</span> degree <span style="color: #8B008B">in</span> <span style="color: #658b00">range</span>(<span style="color: #B452CD">1</span>,Maxpolydegree): <span style="color: #228B22">#No intercept column</span>
|
||||
X[:,degree-<span style="color: #B452CD">1</span>] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style="color: #B452CD">0.2</span>)
|
||||
|
||||
@@ -970,11 +959,13 @@ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=<span style=
|
||||
|
||||
<span style="color: #228B22">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean = np.mean(X_train,axis=<span style="color: #B452CD">0</span>)
|
||||
X_train_scaled = X_train - X_train_mean <span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
<span style="color: #228B22">#Center by removing mean from each feature</span>
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) <span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled = y_train - y_scaler <span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
<span style="color: #228B22">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
<span style="color: #228B22">#Remove the intercept from the training data.</span>
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
|
||||
p = Maxpolydegree-<span style="color: #B452CD">1</span>
|
||||
@@ -989,25 +980,20 @@ lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean<span style="color: #707a7c">@OwnRidgeBeta</span> <span style="color: #228B22">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #228B22">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_
|
||||
<span style="color: #228B22">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler <span style="color: #228B22">#Add intercept to prediction</span>
|
||||
<span style="color: #228B22">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #658b00">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #658b00">print</span>(OwnRidgeBeta) <span style="color: #228B22">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
@@ -1017,14 +1003,10 @@ lambdas = np.logspace(-<span style="color: #B452CD">4</span>, <span style="color
|
||||
<span style="color: #658b00">print</span>(<span style="color: #CD5555">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #658b00">print</span>(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #228B22"># Now plot the results</span>
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, <span style="color: #CD5555">'b--'</span>, label = <span style="color: #CD5555">'MSE own Ridge Test'</span>)
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, <span style="color: #CD5555">'g--'</span>, label = <span style="color: #CD5555">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt.xlabel(<span style="color: #CD5555">'log10(lambda)'</span>)
|
||||
plt.ylabel(<span style="color: #CD5555">'MSE'</span>)
|
||||
plt.legend()
|
||||
|
||||
@@ -99,10 +99,13 @@ div { text-align: justify; text-justify: inter-word; }
|
||||
2,
|
||||
None,
|
||||
'linear-regression-code-intercept-handling-first'),
|
||||
('What does centering mean mathematically?',
|
||||
('What does centering (subtracting the mean values) mean '
|
||||
'mathematically?',
|
||||
2,
|
||||
None,
|
||||
'what-does-centering-mean-mathematically'),
|
||||
'what-does-centering-subtracting-the-mean-values-mean-mathematically'),
|
||||
('Code Examples', 2, None, 'code-examples'),
|
||||
('Taking out the mean', 2, None, 'taking-out-the-mean'),
|
||||
('More complicated Example: The Ising model',
|
||||
2,
|
||||
None,
|
||||
@@ -621,14 +624,16 @@ Furthermore, in for example Ridge and Lasso regression, the solutions
|
||||
If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix \( \boldsymbol{X} \) by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation.
|
||||
standard deviation. Most machine learning libraries do this as a deafult. This means that if you compare your code with the results from a given library,
|
||||
the results may differ. Tracing back the differences may often lead to an increased confusion.
|
||||
|
||||
<p>
|
||||
The
|
||||
<a href="https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html" target="_blank">Standadscaler</a>
|
||||
function in <b>Scikit-Learn</b> does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them.
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
|
||||
<p>
|
||||
If you need to scale the data, not doing so will give an <em>unfair</em>
|
||||
@@ -677,7 +682,7 @@ y_pred <span style="color: #666666">=</span> y_pred <span style="color: #666666"
|
||||
<h2 id="linear-regression-code-intercept-handling-first">Linear Regression code, Intercept handling first </h2>
|
||||
|
||||
<p>
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>)
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (<em>code example thanks to Øyvind Sigmundson Schøyen</em>). Here our scaling of the data is done by subtracting the mean values only.
|
||||
|
||||
<p>
|
||||
|
||||
@@ -756,81 +761,72 @@ plt<span style="color: #666666">.</span>show()
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="what-does-centering-mean-mathematically">What does centering mean mathematically? </h2>
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
<h2 id="what-does-centering-subtracting-the-mean-values-mean-mathematically">What does centering (subtracting the mean values) mean mathematically? </h2>
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is:
|
||||
Let us try to understand what this may imply mathematically when we subtract the mean values, also known as <em>zero centering</em>. To catch many birds with just one stone, we will focus on Ridge regression.
|
||||
|
||||
<p>
|
||||
The cost/loss function for Ridge regression is
|
||||
$$
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2 + \lambda \sum_{j=1}^{p-1} \beta_i^2.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Notice that the intercept is left out of the \( L_2 \) regularization term. The design matrix
|
||||
Note that the intercept term $\beta_0$is left out of the \( L_2 \) regularization term. The design matrix
|
||||
\( X \) does in this case not contain any intercept column. We want
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
$$
|
||||
|
||||
<p>
|
||||
for all \( j \), so lets start with \( \beta_0 \). This means that we have
|
||||
for all \( j \), so let us start with \( \beta_0 \). This means that we have
|
||||
|
||||
$$
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
$$
|
||||
|
||||
<p>
|
||||
We want to solve
|
||||
$$
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
\frac{\partial C}{\partial \beta_0} = -2\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right),
|
||||
$$
|
||||
|
||||
<p>
|
||||
which gives
|
||||
$$
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
$$
|
||||
|
||||
<p>
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
<p>
|
||||
If we assume that every column of \( X \) is centered, whic we can do by subtracting the mean,
|
||||
If we assume that every column of \( \boldsymbol{X} \) is centered, which we can do by subtracting the mean,
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span>X <span style="color: #666666">=</span> X <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(X,axis<span style="color: #666666">=0</span>)
|
||||
</pre></div>
|
||||
<p>
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
<p>
|
||||
the sum \( \sum_{i=0}^{n-1} X_{ij} \)
|
||||
can be rewritten as
|
||||
$$
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
\sum_{i=0}^{n-1} \left(X_{ij} - \frac{1}{n}\sum_{i=0}^{n-1} X_{ij}) = \sum_{i=0}^{n-1} X_{ij} - \sum_{i=0}^{n-1} \frac{1}{n} \sum_{i=0}^{n-1}X_{ij},
|
||||
$$
|
||||
|
||||
resulting in
|
||||
$$
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
\sum_{i=0}^{n-1} X_{ij} - n \frac{1}{n} \sum_{i=0}^{n-1}X_{ij} = 0.
|
||||
$$
|
||||
|
||||
<p>
|
||||
Finally we have
|
||||
$$
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
n\beta_0 = \sum_{i=0}^{n-1} y_i - \sum_{j=1}^{p-1}\beta_j \sum_{i=0}^{n-1} X_{ij},
|
||||
$$
|
||||
|
||||
or
|
||||
$$
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1} y_i = \overline{\boldsymbol{y}},
|
||||
$$
|
||||
|
||||
the average value of \( \boldsymbol{y] \).
|
||||
|
||||
<p>
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - y_{average} \) in the loss function will give us (in vector-matrix disguise)
|
||||
Replacing \( y_i \) with \( y_i - \beta_0 = y_i - \overline{\boldsymbol{y}} \) in the cost function will give us (in vector-matrix disguise)
|
||||
$$
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
$$
|
||||
@@ -840,9 +836,17 @@ which has the solution
|
||||
|
||||
<p>
|
||||
\( \beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}} \).
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average} \)
|
||||
where \( \boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\boldsymbol{y}} \)
|
||||
and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="code-examples">Code Examples </h2>
|
||||
|
||||
<p>
|
||||
Armed with this wisdom, we attempt first simply set the intercept eqault to <b>False</b> in our implementation of Ridge regression for a vanilla data set.
|
||||
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
@@ -852,8 +856,6 @@ and \( \tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj} \).
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.model_selection</span> <span style="color: #008000; font-weight: bold">import</span> train_test_split
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">R2</span>(y_data, y_model):
|
||||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span> <span style="color: #666666">-</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> y_model) <span style="color: #666666">**</span> <span style="color: #666666">2</span>) <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(y_data)) <span style="color: #666666">**</span> <span style="color: #666666">2</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
@@ -869,42 +871,29 @@ y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>e
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree))
|
||||
We include explicitely the intercpt column
|
||||
X[:,<span style="color: #666666">0</span>] <span style="color: #666666">=</span> <span style="color: #666666">1.0</span>
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> polydegree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>, Maxpolydegree):
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(polydegree):
|
||||
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(Maxpolydegree):
|
||||
X[:,degree] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>degree
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
<span style="color: #408080; font-style: italic"># matrix inversion to find beta</span>
|
||||
OLSbeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #008000">print</span>(OLSbeta)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
ytildeOLS <span style="color: #666666">=</span> X_train <span style="color: #666666">@</span> OLSbeta
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Training MSE for OLS"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y_train,ytildeOLS))
|
||||
ypredictOLS <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OLSbeta
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Test MSE OLS"</span>)
|
||||
<span style="color: #008000">print</span>(MSE(y_test,ypredictOLS))
|
||||
|
||||
p <span style="color: #666666">=</span> <span style="color: #008000">len</span>(OLSbeta)
|
||||
p <span style="color: #666666">=</span> Maxpolydegree
|
||||
I <span style="color: #666666">=</span> np<span style="color: #666666">.</span>eye(p,p)
|
||||
<span style="color: #408080; font-style: italic"># Decide which values of lambda to use</span>
|
||||
nlambdas <span style="color: #666666">=</span> <span style="color: #666666">4</span>
|
||||
MSEOwnRidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSEOwnRidgeTrain <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgePredict <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
MSERidgeTrain <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros(nlambdas)
|
||||
|
||||
lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</span>logspace(<span style="color: #666666">-4</span>, <span style="color: #666666">4</span>, nlambdas)
|
||||
<span style="color: #008000; font-weight: bold">for</span> i <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(nlambdas):
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train<span style="color: #666666">.</span>T <span style="color: #666666">@</span> y_train
|
||||
<span style="color: #408080; font-style: italic"># include lasso using Scikit-Learn</span>
|
||||
<span style="color: #408080; font-style: italic"># Note: we include the intercept</span>
|
||||
<span style="color: #408080; font-style: italic"># Note: we include the intercept column and no scaling</span>
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb,fit_intercept<span style="color: #666666">=</span><span style="color: #008000; font-weight: bold">False</span>)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
<span style="color: #408080; font-style: italic"># and then make the prediction</span>
|
||||
@@ -913,18 +902,14 @@ lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</
|
||||
ytildeRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSEOwnRidgeTrain[i] <span style="color: #666666">=</span> MSE(y_train,ytildeOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
MSERidgeTrain[i] <span style="color: #666666">=</span> MSE(y_train,ytildeRidge)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>coef_)
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgeTrain, <span style="color: #BA2121">'b'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge train'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'r'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgeTrain, <span style="color: #BA2121">'y'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge train'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
@@ -933,6 +918,17 @@ plt<span style="color: #666666">.</span>legend()
|
||||
plt<span style="color: #666666">.</span>show()
|
||||
</pre></div>
|
||||
<p>
|
||||
The results here agree when we force <b>Scikit-Learn</b>'s Ridge function to include the first column in our design matrix.
|
||||
The problem however is that can easily lead to a larger mean-squared error!
|
||||
|
||||
<p>
|
||||
Let us see how we can change this code by zero centering.
|
||||
|
||||
<p>
|
||||
<!-- !split --><br><br><br><br><br><br><br><br><br><br>
|
||||
|
||||
<h2 id="taking-out-the-mean">Taking out the mean </h2>
|
||||
<p>
|
||||
|
||||
<!-- code=python (!bc pycod) typeset with pygments style "default" -->
|
||||
<div class="highlight" style="background: #f8f8f8"><pre style="line-height: 125%;"><span></span><span style="color: #008000; font-weight: bold">import</span> <span style="color: #0000FF; font-weight: bold">numpy</span> <span style="color: #008000; font-weight: bold">as</span> <span style="color: #0000FF; font-weight: bold">np</span>
|
||||
@@ -942,13 +938,9 @@ plt<span style="color: #666666">.</span>show()
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn</span> <span style="color: #008000; font-weight: bold">import</span> linear_model
|
||||
<span style="color: #008000; font-weight: bold">from</span> <span style="color: #0000FF; font-weight: bold">sklearn.preprocessing</span> <span style="color: #008000; font-weight: bold">import</span> StandardScaler
|
||||
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">R2</span>(y_data, y_model):
|
||||
<span style="color: #008000; font-weight: bold">return</span> <span style="color: #666666">1</span> <span style="color: #666666">-</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> y_model) <span style="color: #666666">**</span> <span style="color: #666666">2</span>) <span style="color: #666666">/</span> np<span style="color: #666666">.</span>sum((y_data <span style="color: #666666">-</span> np<span style="color: #666666">.</span>mean(y_data)) <span style="color: #666666">**</span> <span style="color: #666666">2</span>)
|
||||
<span style="color: #008000; font-weight: bold">def</span> <span style="color: #0000FF">MSE</span>(y_data,y_model):
|
||||
n <span style="color: #666666">=</span> np<span style="color: #666666">.</span>size(y_model)
|
||||
<span style="color: #008000; font-weight: bold">return</span> np<span style="color: #666666">.</span>sum((y_data<span style="color: #666666">-</span>y_model)<span style="color: #666666">**2</span>)<span style="color: #666666">/</span>n
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># A seed just to ensure that the random numbers are the same for every run.</span>
|
||||
<span style="color: #408080; font-style: italic"># Useful for eventual debugging.</span>
|
||||
np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>seed(<span style="color: #666666">315</span>)
|
||||
@@ -957,15 +949,12 @@ n <span style="color: #666666">=</span> <span style="color: #666666">100</span>
|
||||
x <span style="color: #666666">=</span> np<span style="color: #666666">.</span>random<span style="color: #666666">.</span>rand(n)
|
||||
y <span style="color: #666666">=</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>x<span style="color: #666666">**2</span>) <span style="color: #666666">+</span> <span style="color: #666666">1.5</span> <span style="color: #666666">*</span> np<span style="color: #666666">.</span>exp(<span style="color: #666666">-</span>(x<span style="color: #666666">-2</span>)<span style="color: #666666">**2</span>)
|
||||
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">5</span>
|
||||
Maxpolydegree <span style="color: #666666">=</span> <span style="color: #666666">20</span>
|
||||
X <span style="color: #666666">=</span> np<span style="color: #666666">.</span>zeros((n,Maxpolydegree<span style="color: #666666">-1</span>))
|
||||
|
||||
<span style="color: #008000; font-weight: bold">for</span> degree <span style="color: #AA22FF; font-weight: bold">in</span> <span style="color: #008000">range</span>(<span style="color: #666666">1</span>,Maxpolydegree): <span style="color: #408080; font-style: italic">#No intercept column</span>
|
||||
X[:,degree<span style="color: #666666">-1</span>] <span style="color: #666666">=</span> x<span style="color: #666666">**</span>(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># We split the data in test and training data</span>
|
||||
X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_test_split(X, y, test_size<span style="color: #666666">=0.2</span>)
|
||||
|
||||
@@ -975,11 +964,13 @@ X_train, X_test, y_train, y_test <span style="color: #666666">=</span> train_tes
|
||||
|
||||
<span style="color: #408080; font-style: italic">#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable</span>
|
||||
X_train_mean <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(X_train,axis<span style="color: #666666">=0</span>)
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean <span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
<span style="color: #408080; font-style: italic">#Center by removing mean from each feature</span>
|
||||
X_train_scaled <span style="color: #666666">=</span> X_train <span style="color: #666666">-</span> X_train_mean
|
||||
X_test_scaled <span style="color: #666666">=</span> X_test <span style="color: #666666">-</span> X_train_mean
|
||||
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train) <span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler <span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
<span style="color: #408080; font-style: italic">#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)</span>
|
||||
<span style="color: #408080; font-style: italic">#Remove the intercept from the training data.</span>
|
||||
y_scaler <span style="color: #666666">=</span> np<span style="color: #666666">.</span>mean(y_train)
|
||||
y_train_scaled <span style="color: #666666">=</span> y_train <span style="color: #666666">-</span> y_scaler
|
||||
|
||||
|
||||
p <span style="color: #666666">=</span> Maxpolydegree<span style="color: #666666">-1</span>
|
||||
@@ -994,25 +985,20 @@ lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</
|
||||
lmb <span style="color: #666666">=</span> lambdas[i]
|
||||
OwnRidgeBeta <span style="color: #666666">=</span> np<span style="color: #666666">.</span>linalg<span style="color: #666666">.</span>pinv(X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">+</span>lmb<span style="color: #666666">*</span>I) <span style="color: #666666">@</span> X_train_scaled<span style="color: #666666">.</span>T <span style="color: #666666">@</span> (y_train_scaled)
|
||||
intercept_ <span style="color: #666666">=</span> y_scaler <span style="color: #666666">-</span> X_train_mean<span style="color: #AA22FF">@OwnRidgeBeta</span> <span style="color: #408080; font-style: italic">#The intercept can be shifted so the model can predict on uncentered data</span>
|
||||
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> intercept_ <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> intercept_
|
||||
<span style="color: #408080; font-style: italic">#EQUIVALENT PREDICTION:</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler <span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
<span style="color: #408080; font-style: italic">#Add intercept to prediction</span>
|
||||
ypredictOwnRidge <span style="color: #666666">=</span> X_test_scaled <span style="color: #666666">@</span> OwnRidgeBeta <span style="color: #666666">+</span> y_scaler
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for own Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge <span style="color: #666666">=</span> linear_model<span style="color: #666666">.</span>Ridge(lmb)
|
||||
RegRidge<span style="color: #666666">.</span>fit(X_train,y_train)
|
||||
ypredictRidge <span style="color: #666666">=</span> RegRidge<span style="color: #666666">.</span>predict(X_test)
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Values for SL Ridge prediction"</span>)
|
||||
<span style="color: #008000">print</span>(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] <span style="color: #666666">=</span> MSE(y_test,ypredictRidge)
|
||||
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for own Ridge implementation"</span>)
|
||||
<span style="color: #008000">print</span>(OwnRidgeBeta) <span style="color: #408080; font-style: italic">#Intercept is given by mean of target variable</span>
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">"Beta values for Scikit-Learn Ridge implementation"</span>)
|
||||
@@ -1022,14 +1008,10 @@ lambdas <span style="color: #666666">=</span> np<span style="color: #666666">.</
|
||||
<span style="color: #008000">print</span>(<span style="color: #BA2121">'Intercept from Scikit-Learn Ridge implementation'</span>)
|
||||
<span style="color: #008000">print</span>(RegRidge<span style="color: #666666">.</span>intercept_)
|
||||
|
||||
|
||||
|
||||
<span style="color: #408080; font-style: italic"># Now plot the results</span>
|
||||
|
||||
plt<span style="color: #666666">.</span>figure()
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSEOwnRidgePredict, <span style="color: #BA2121">'b--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE own Ridge Test'</span>)
|
||||
plt<span style="color: #666666">.</span>plot(np<span style="color: #666666">.</span>log10(lambdas), MSERidgePredict, <span style="color: #BA2121">'g--'</span>, label <span style="color: #666666">=</span> <span style="color: #BA2121">'MSE SL Ridge Test'</span>)
|
||||
|
||||
plt<span style="color: #666666">.</span>xlabel(<span style="color: #BA2121">'log10(lambda)'</span>)
|
||||
plt<span style="color: #666666">.</span>ylabel(<span style="color: #BA2121">'MSE'</span>)
|
||||
plt<span style="color: #666666">.</span>legend()
|
||||
|
||||
Binary file not shown.
@@ -382,13 +382,15 @@
|
||||
"If our predictors represent different scales, then it is important to\n",
|
||||
"standardize the design matrix $\\boldsymbol{X}$ by subtracting the mean of each\n",
|
||||
"column from the corresponding column and dividing the column with its\n",
|
||||
"standard deviation.\n",
|
||||
"standard deviation. Most machine learning libraries do this as a deafult. This means that if you compare your code with the results from a given library,\n",
|
||||
"the results may differ. Tracing back the differences may often lead to an increased confusion.\n",
|
||||
"\n",
|
||||
"The\n",
|
||||
"[Standadscaler](https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html)\n",
|
||||
"function in **Scikit-Learn** does this for us. For the data sets we\n",
|
||||
"have been studying in our various examples, the data are in many cases\n",
|
||||
"already scaled and there is no need to scale them.\n",
|
||||
"already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a\n",
|
||||
"survey of your data, with a critical assessment of them in case you need to scale the data.\n",
|
||||
"\n",
|
||||
"If you need to scale the data, not doing so will give an *unfair*\n",
|
||||
"penalization of the parameters since their magnitude depends on the\n",
|
||||
@@ -440,7 +442,7 @@
|
||||
"source": [
|
||||
"## Linear Regression code, Intercept handling first\n",
|
||||
"\n",
|
||||
"This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*)"
|
||||
"This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*). Here our scaling of the data is done by subtracting the mean values only."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -528,13 +530,12 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"## What does centering mean mathematically?\n",
|
||||
"Here is a mathematical explanation of the zero centering:\n",
|
||||
"## What does centering (subtracting the mean values) mean mathematically?\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"Let us try to understand what this may imply mathematically when we subtract the mean values, also known as *zero centering*. To catch many birds with just one stone, we will focus on Ridge regression.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"The cost/loss function for Ridge regression is:"
|
||||
"The cost/loss function for Ridge regression is"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -542,7 +543,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"C(\\beta_0, \\beta_1, ... , \\beta_P) = \\sum_{i=1}^{n} (y_i - \\beta_0 - \\sum_{p=1}^P X_{ip}\\beta_p)^2 + \\lambda \\sum_{p=1}^P \\beta_p^2.\n",
|
||||
"C(\\beta_0, \\beta_1, ... , \\beta_{p-1}) = \\sum_{i=0}^{n} \\left(y_i - \\beta_0 - \\sum_{j=1}^{p-1} X_{ij}\\beta_j\\right)^2 + \\lambda \\sum_{j=1}^{p-1} \\beta_i^2.\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -550,7 +551,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"Notice that the intercept is left out of the $L_2$ regularization term. The design matrix\n",
|
||||
"Note that the intercept term $\\beta_0$is left out of the $L_2$ regularization term. The design matrix\n",
|
||||
"$X$ does in this case not contain any intercept column. We want"
|
||||
]
|
||||
},
|
||||
@@ -559,7 +560,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\frac{\\partial L}{\\partial \\beta_j} = 0,\n",
|
||||
"\\frac{\\partial C}{\\partial \\beta_j} = 0,\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -567,7 +568,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"for all $j$, so lets start with $\\beta_0$. This means that we have"
|
||||
"for all $j$, so let us start with $\\beta_0$. This means that we have"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -575,23 +576,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\frac{\\partial L}{\\partial \\beta_0} = -2\\sum_{i=1}^{n} (y_i - \\beta_0 - \\sum_{p=1}^P X_{ip} \\beta_p).\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"We want to solve"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"-2\\sum_{i=1}^{n} (y_i - \\beta_0 - \\sum_{p=1}^P X_{ip} \\beta_p) = 0,\n",
|
||||
"\\frac{\\partial C}{\\partial \\beta_0} = -2\\sum_{i=0}^{n-1} \\left(y_i - \\beta_0 - \\sum_{j=1}^{p-1} X_{ij} \\beta_j\\right),\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -607,7 +592,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\sum_{i=1}^{n} \\beta_0 = \\sum_{i=1}^{n}y_i - \\sum_{i=1}^{n} \\sum_{p=1}^P X_{ip} \\beta_p,\n",
|
||||
"\\sum_{i=0}^{n-1} \\beta_0 = \\sum_{i=0}^{n-1}y_i - \\sum_{i=0}^{n-1} \\sum_{j=1}^{p-1} X_{ij} \\beta_j.\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -615,13 +600,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"or\n",
|
||||
"$ n\\beta_0 = \\sum_{i=1}^{n} y_i - \\sum_{p=1}^P\\beta_p \\sum_{i=1}^{n} X_{ip}$.\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"If we assume that every column of $X$ is centered, whic we can do by subtracting the mean,"
|
||||
"If we assume that every column of $\\boldsymbol{X}$ is centered, which we can do by subtracting the mean,"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -640,8 +619,7 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"the sum $ \\sum_{i=1}^{n} X_{ip} $\n",
|
||||
"\n",
|
||||
"the sum $\\sum_{i=0}^{n-1} X_{ij}$\n",
|
||||
"can be rewritten as"
|
||||
]
|
||||
},
|
||||
@@ -650,7 +628,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\sum_{i=1}^{n} (X_{ip} - \\frac{1}{n}\\sum_{i=1}^{n} X_{ip}) = \\sum_{i=1}^{n} X_{ip} - \\sum_{i=1}^{n} \\frac{1}{n} \\sum_{i=1}^{n}X_{ip},\n",
|
||||
"\\sum_{i=0}^{n-1} \\left(X_{ij} - \\frac{1}{n}\\sum_{i=0}^{n-1} X_{ij}) = \\sum_{i=0}^{n-1} X_{ij} - \\sum_{i=0}^{n-1} \\frac{1}{n} \\sum_{i=0}^{n-1}X_{ij},\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -666,7 +644,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\sum_{i=1}^{n} X_{ip} - n \\frac{1}{n} \\sum_{i=1}^{n}X_{ip} = 0.\n",
|
||||
"\\sum_{i=0}^{n-1} X_{ij} - n \\frac{1}{n} \\sum_{i=0}^{n-1}X_{ij} = 0.\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -682,7 +660,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"n\\beta_0 = \\sum_{i=1}^{n} y_i - \\sum_{p=1}^P\\beta_p \\sum_{i=1}^{n} X_{ip},\n",
|
||||
"n\\beta_0 = \\sum_{i=0}^{n-1} y_i - \\sum_{j=1}^{p-1}\\beta_j \\sum_{i=0}^{n-1} X_{ij},\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -698,7 +676,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"$$\n",
|
||||
"\\beta_0 = \\frac{1}{n}\\sum_{i=1}^{n} y_i = y_{average}.\n",
|
||||
"\\beta_0 = \\frac{1}{n}\\sum_{i=0}^{n-1} y_i = \\overline{\\boldsymbol{y}},\n",
|
||||
"$$"
|
||||
]
|
||||
},
|
||||
@@ -706,7 +684,9 @@
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"Replacing $y_i$ with $y_i - \\beta_0 = y_i - y_{average}$ in the loss function will give us (in vector-matrix disguise)"
|
||||
"the average value of $\\boldsymbol{y]$.\n",
|
||||
"\n",
|
||||
"Replacing $y_i$ with $y_i - \\beta_0 = y_i - \\overline{\\boldsymbol{y}}$ in the cost function will give us (in vector-matrix disguise)"
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -725,8 +705,12 @@
|
||||
"which has the solution\n",
|
||||
"\n",
|
||||
"$\\beta = (\\tilde{X}^T\\tilde{X} + \\lambda I)^{-1}\\tilde{X}^T\\boldsymbol{\\tilde{y}}$.\n",
|
||||
"where $\\boldsymbol{\\tilde{y}} = \\boldsymbol{y} - y_{average}$\n",
|
||||
"and $\\tilde{X}_{ij} = X_{ij} - \\frac{1}{n}\\sum_{k=1}^{n-1}X_{kj}$."
|
||||
"where $\\boldsymbol{\\tilde{y}} = \\boldsymbol{y} - \\overline{\\boldsymbol{y}}$\n",
|
||||
"and $\\tilde{X}_{ij} = X_{ij} - \\frac{1}{n}\\sum_{k=1}^{n-1}X_{kj}$.\n",
|
||||
"\n",
|
||||
"## Code Examples\n",
|
||||
"\n",
|
||||
"Armed with this wisdom, we attempt first simply set the intercept eqault to **False** in our implementation of Ridge regression for a vanilla data set."
|
||||
]
|
||||
},
|
||||
{
|
||||
@@ -744,8 +728,6 @@
|
||||
"from sklearn.model_selection import train_test_split\n",
|
||||
"from sklearn import linear_model\n",
|
||||
"\n",
|
||||
"def R2(y_data, y_model):\n",
|
||||
" return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n",
|
||||
"def MSE(y_data,y_model):\n",
|
||||
" n = np.size(y_model)\n",
|
||||
" return np.sum((y_data-y_model)**2)/n\n",
|
||||
@@ -761,42 +743,29 @@
|
||||
"\n",
|
||||
"Maxpolydegree = 20\n",
|
||||
"X = np.zeros((n,Maxpolydegree))\n",
|
||||
"We include explicitely the intercpt column\n",
|
||||
"X[:,0] = 1.0\n",
|
||||
"\n",
|
||||
"for polydegree in range(1, Maxpolydegree):\n",
|
||||
" for degree in range(polydegree):\n",
|
||||
" X[:,degree] = x**degree\n",
|
||||
"for degree in range(Maxpolydegree):\n",
|
||||
" X[:,degree] = x**degree\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# We split the data in test and training data\n",
|
||||
"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n",
|
||||
"\n",
|
||||
"# matrix inversion to find beta\n",
|
||||
"OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train\n",
|
||||
"print(OLSbeta)\n",
|
||||
"# and then make the prediction\n",
|
||||
"ytildeOLS = X_train @ OLSbeta\n",
|
||||
"print(\"Training MSE for OLS\")\n",
|
||||
"print(MSE(y_train,ytildeOLS))\n",
|
||||
"ypredictOLS = X_test @ OLSbeta\n",
|
||||
"print(\"Test MSE OLS\")\n",
|
||||
"print(MSE(y_test,ypredictOLS))\n",
|
||||
"\n",
|
||||
"p = len(OLSbeta)\n",
|
||||
"p = Maxpolydegree\n",
|
||||
"I = np.eye(p,p)\n",
|
||||
"# Decide which values of lambda to use\n",
|
||||
"nlambdas = 4\n",
|
||||
"MSEOwnRidgePredict = np.zeros(nlambdas)\n",
|
||||
"MSEOwnRidgeTrain = np.zeros(nlambdas)\n",
|
||||
"MSERidgePredict = np.zeros(nlambdas)\n",
|
||||
"MSERidgeTrain = np.zeros(nlambdas)\n",
|
||||
"\n",
|
||||
"lambdas = np.logspace(-4, 4, nlambdas)\n",
|
||||
"for i in range(nlambdas):\n",
|
||||
" lmb = lambdas[i]\n",
|
||||
" OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train\n",
|
||||
" # include lasso using Scikit-Learn\n",
|
||||
" # Note: we include the intercept\n",
|
||||
" # Note: we include the intercept column and no scaling\n",
|
||||
" RegRidge = linear_model.Ridge(lmb,fit_intercept=False)\n",
|
||||
" RegRidge.fit(X_train,y_train)\n",
|
||||
" # and then make the prediction\n",
|
||||
@@ -805,18 +774,14 @@
|
||||
" ytildeRidge = RegRidge.predict(X_train)\n",
|
||||
" ypredictRidge = RegRidge.predict(X_test)\n",
|
||||
" MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n",
|
||||
" MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)\n",
|
||||
" MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n",
|
||||
" MSERidgeTrain[i] = MSE(y_train,ytildeRidge)\n",
|
||||
" print(\"Beta values for own Ridge implementation\")\n",
|
||||
" print(OwnRidgeBeta)\n",
|
||||
" print(\"Beta values for Scikit-Learn Ridge implementation\")\n",
|
||||
" print(RegRidge.coef_)\n",
|
||||
"# Now plot the results\n",
|
||||
"plt.figure()\n",
|
||||
"plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, 'b', label = 'MSE Ridge train')\n",
|
||||
"plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE Ridge Test')\n",
|
||||
"plt.plot(np.log10(lambdas), MSERidgeTrain, 'y', label = 'MSE Ridge train')\n",
|
||||
"plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')\n",
|
||||
"\n",
|
||||
"plt.xlabel('log10(lambda)')\n",
|
||||
@@ -825,6 +790,18 @@
|
||||
"plt.show()"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "markdown",
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"The results here agree when we force **Scikit-Learn**'s Ridge function to include the first column in our design matrix.\n",
|
||||
"The problem however is that can easily lead to a larger mean-squared error!\n",
|
||||
"\n",
|
||||
"Let us see how we can change this code by zero centering.\n",
|
||||
"\n",
|
||||
"## Taking out the mean"
|
||||
]
|
||||
},
|
||||
{
|
||||
"cell_type": "code",
|
||||
"execution_count": null,
|
||||
@@ -841,13 +818,9 @@
|
||||
"from sklearn import linear_model\n",
|
||||
"from sklearn.preprocessing import StandardScaler\n",
|
||||
"\n",
|
||||
"def R2(y_data, y_model):\n",
|
||||
" return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)\n",
|
||||
"def MSE(y_data,y_model):\n",
|
||||
" n = np.size(y_model)\n",
|
||||
" return np.sum((y_data-y_model)**2)/n\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# A seed just to ensure that the random numbers are the same for every run.\n",
|
||||
"# Useful for eventual debugging.\n",
|
||||
"np.random.seed(315)\n",
|
||||
@@ -856,15 +829,12 @@
|
||||
"x = np.random.rand(n)\n",
|
||||
"y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)\n",
|
||||
"\n",
|
||||
"Maxpolydegree = 5\n",
|
||||
"Maxpolydegree = 20\n",
|
||||
"X = np.zeros((n,Maxpolydegree-1))\n",
|
||||
"\n",
|
||||
"for degree in range(1,Maxpolydegree): #No intercept column\n",
|
||||
" X[:,degree-1] = x**(degree)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# We split the data in test and training data\n",
|
||||
"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)\n",
|
||||
"\n",
|
||||
@@ -874,11 +844,13 @@
|
||||
"\n",
|
||||
"#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable\n",
|
||||
"X_train_mean = np.mean(X_train,axis=0)\n",
|
||||
"X_train_scaled = X_train - X_train_mean #Center by removing mean from each feature\n",
|
||||
"#Center by removing mean from each feature\n",
|
||||
"X_train_scaled = X_train - X_train_mean \n",
|
||||
"X_test_scaled = X_test - X_train_mean\n",
|
||||
"\n",
|
||||
"y_scaler = np.mean(y_train) #The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)\n",
|
||||
"y_train_scaled = y_train - y_scaler #Remove the intercept from the training data.\n",
|
||||
"#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)\n",
|
||||
"#Remove the intercept from the training data.\n",
|
||||
"y_scaler = np.mean(y_train) \n",
|
||||
"y_train_scaled = y_train - y_scaler \n",
|
||||
"\n",
|
||||
"\n",
|
||||
"p = Maxpolydegree-1\n",
|
||||
@@ -893,25 +865,20 @@
|
||||
" lmb = lambdas[i]\n",
|
||||
" OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)\n",
|
||||
" intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data\n",
|
||||
" \n",
|
||||
" ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ #Add intercept to prediction\n",
|
||||
" #Add intercept to prediction\n",
|
||||
" ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ \n",
|
||||
" #EQUIVALENT PREDICTION:\n",
|
||||
" ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler #Add intercept to prediction\n",
|
||||
" #Add intercept to prediction\n",
|
||||
" ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler \n",
|
||||
" print(\"Values for own Ridge prediction\")\n",
|
||||
" print(ypredictOwnRidge)\n",
|
||||
"\n",
|
||||
" \n",
|
||||
"\n",
|
||||
" RegRidge = linear_model.Ridge(lmb)\n",
|
||||
" RegRidge.fit(X_train,y_train)\n",
|
||||
" ypredictRidge = RegRidge.predict(X_test)\n",
|
||||
" print(\"Values for SL Ridge prediction\")\n",
|
||||
" print(ypredictRidge)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
" MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)\n",
|
||||
" MSERidgePredict[i] = MSE(y_test,ypredictRidge)\n",
|
||||
"\n",
|
||||
" print(\"Beta values for own Ridge implementation\")\n",
|
||||
" print(OwnRidgeBeta) #Intercept is given by mean of target variable\n",
|
||||
" print(\"Beta values for Scikit-Learn Ridge implementation\")\n",
|
||||
@@ -921,14 +888,10 @@
|
||||
" print('Intercept from Scikit-Learn Ridge implementation')\n",
|
||||
" print(RegRidge.intercept_)\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"# Now plot the results\n",
|
||||
"\n",
|
||||
"plt.figure()\n",
|
||||
"plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')\n",
|
||||
"plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')\n",
|
||||
"\n",
|
||||
"plt.xlabel('log10(lambda)')\n",
|
||||
"plt.ylabel('MSE')\n",
|
||||
"plt.legend()\n",
|
||||
@@ -1603,7 +1566,7 @@
|
||||
"metadata": {},
|
||||
"source": [
|
||||
"3\n",
|
||||
"2\n",
|
||||
"1\n",
|
||||
" \n",
|
||||
"<\n",
|
||||
"<\n",
|
||||
|
||||
@@ -271,13 +271,15 @@ $\bm{X}$ are zero centered, that is we subtract the mean values.
|
||||
If our predictors represent different scales, then it is important to
|
||||
standardize the design matrix $\bm{X}$ by subtracting the mean of each
|
||||
column from the corresponding column and dividing the column with its
|
||||
standard deviation.
|
||||
standard deviation. Most machine learning libraries do this as a deafult. This means that if you compare your code with the results from a given library,
|
||||
the results may differ. Tracing back the differences may often lead to an increased confusion.
|
||||
|
||||
The
|
||||
"Standadscaler":"https://scikit-learn.org/stable/modules/generated/sklearn.preprocessing.StandardScaler.html"
|
||||
function in _Scikit-Learn_ does this for us. For the data sets we
|
||||
have been studying in our various examples, the data are in many cases
|
||||
already scaled and there is no need to scale them.
|
||||
already scaled and there is no need to scale them. You as a user of different machine learning algorithms, should always perform a
|
||||
survey of your data, with a critical assessment of them in case you need to scale the data.
|
||||
|
||||
If you need to scale the data, not doing so will give an *unfair*
|
||||
penalization of the parameters since their magnitude depends on the
|
||||
@@ -318,7 +320,7 @@ y_pred = y_pred + y_train_mean
|
||||
!split
|
||||
===== Linear Regression code, Intercept handling first =====
|
||||
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*)
|
||||
This code shows a simple first-order fit to a data set using the above transformed data, where we consider the role of the intercept first, by either excluding it or including it (*code example thanks to Øyvind Sigmundson Schøyen*). Here our scaling of the data is done by subtracting the mean values only.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
@@ -395,78 +397,57 @@ plt.show()
|
||||
!ec
|
||||
|
||||
!split
|
||||
===== What does centering mean mathematically? =====
|
||||
Here is a mathematical explanation of the zero centering:
|
||||
===== What does centering (subtracting the mean values) mean mathematically? =====
|
||||
|
||||
|
||||
Let us try to understand what this may imply mathematically when we subtract the mean values, also known as *zero centering*. To catch many birds with just one stone, we will focus on Ridge regression.
|
||||
|
||||
|
||||
The cost/loss function for Ridge regression is:
|
||||
|
||||
|
||||
The cost/loss function for Ridge regression is
|
||||
!bt
|
||||
\[
|
||||
C(\beta_0, \beta_1, ... , \beta_P) = \sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip}\beta_p)^2 + \lambda \sum_{p=1}^P \beta_p^2.
|
||||
C(\beta_0, \beta_1, ... , \beta_{p-1}) = \sum_{i=0}^{n} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij}\beta_j\right)^2 + \lambda \sum_{j=1}^{p-1} \beta_i^2.
|
||||
\]
|
||||
!et
|
||||
|
||||
|
||||
|
||||
|
||||
Notice that the intercept is left out of the $L_2$ regularization term. The design matrix
|
||||
Note that the intercept term $\beta_0$is left out of the $L_2$ regularization term. The design matrix
|
||||
$X$ does in this case not contain any intercept column. We want
|
||||
|
||||
!bt
|
||||
\[
|
||||
\frac{\partial L}{\partial \beta_j} = 0,
|
||||
\frac{\partial C}{\partial \beta_j} = 0,
|
||||
\]
|
||||
!et
|
||||
|
||||
for all $j$, so lets start with $\beta_0$. This means that we have
|
||||
for all $j$, so let us start with $\beta_0$. This means that we have
|
||||
|
||||
!bt
|
||||
\[
|
||||
\frac{\partial L}{\partial \beta_0} = -2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p).
|
||||
\frac{\partial C}{\partial \beta_0} = -2\sum_{i=0}^{n-1} \left(y_i - \beta_0 - \sum_{j=1}^{p-1} X_{ij} \beta_j\right),
|
||||
\]
|
||||
!et
|
||||
|
||||
We want to solve
|
||||
!bt
|
||||
\[
|
||||
-2\sum_{i=1}^{n} (y_i - \beta_0 - \sum_{p=1}^P X_{ip} \beta_p) = 0,
|
||||
\]
|
||||
!et
|
||||
|
||||
|
||||
which gives
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=1}^{n} \beta_0 = \sum_{i=1}^{n}y_i - \sum_{i=1}^{n} \sum_{p=1}^P X_{ip} \beta_p,
|
||||
\sum_{i=0}^{n-1} \beta_0 = \sum_{i=0}^{n-1}y_i - \sum_{i=0}^{n-1} \sum_{j=1}^{p-1} X_{ij} \beta_j.
|
||||
\]
|
||||
!et
|
||||
|
||||
or
|
||||
$ n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip}$.
|
||||
|
||||
|
||||
|
||||
|
||||
If we assume that every column of $X$ is centered, whic we can do by subtracting the mean,
|
||||
If we assume that every column of $\bm{X}$ is centered, which we can do by subtracting the mean,
|
||||
!bc pycod
|
||||
X = X - np.mean(X,axis=0)
|
||||
!ec
|
||||
the sum $ \sum_{i=1}^{n} X_{ip} $
|
||||
|
||||
the sum $\sum_{i=0}^{n-1} X_{ij}$
|
||||
can be rewritten as
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=1}^{n} (X_{ip} - \frac{1}{n}\sum_{i=1}^{n} X_{ip}) = \sum_{i=1}^{n} X_{ip} - \sum_{i=1}^{n} \frac{1}{n} \sum_{i=1}^{n}X_{ip},
|
||||
\sum_{i=0}^{n-1} \left(X_{ij} - \frac{1}{n}\sum_{i=0}^{n-1} X_{ij}) = \sum_{i=0}^{n-1} X_{ij} - \sum_{i=0}^{n-1} \frac{1}{n} \sum_{i=0}^{n-1}X_{ij},
|
||||
\]
|
||||
!et
|
||||
resulting in
|
||||
!bt
|
||||
\[
|
||||
\sum_{i=1}^{n} X_{ip} - n \frac{1}{n} \sum_{i=1}^{n}X_{ip} = 0.
|
||||
\sum_{i=0}^{n-1} X_{ij} - n \frac{1}{n} \sum_{i=0}^{n-1}X_{ij} = 0.
|
||||
\]
|
||||
!et
|
||||
|
||||
@@ -474,17 +455,18 @@ resulting in
|
||||
Finally we have
|
||||
!bt
|
||||
\[
|
||||
n\beta_0 = \sum_{i=1}^{n} y_i - \sum_{p=1}^P\beta_p \sum_{i=1}^{n} X_{ip},
|
||||
n\beta_0 = \sum_{i=0}^{n-1} y_i - \sum_{j=1}^{p-1}\beta_j \sum_{i=0}^{n-1} X_{ij},
|
||||
\]
|
||||
!et
|
||||
or
|
||||
!bt
|
||||
\[
|
||||
\beta_0 = \frac{1}{n}\sum_{i=1}^{n} y_i = y_{average}.
|
||||
\beta_0 = \frac{1}{n}\sum_{i=0}^{n-1} y_i = \overline{\bm{y}},
|
||||
\]
|
||||
!et
|
||||
the average value of $\bm{y]$.
|
||||
|
||||
Replacing $y_i$ with $y_i - \beta_0 = y_i - y_{average}$ in the loss function will give us (in vector-matrix disguise)
|
||||
Replacing $y_i$ with $y_i - \beta_0 = y_i - \overline{\bm{y}}$ in the cost function will give us (in vector-matrix disguise)
|
||||
!bt
|
||||
\[
|
||||
C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T(\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta}) + \lambda \boldsymbol{\beta}^T\boldsymbol{\beta},
|
||||
@@ -494,11 +476,13 @@ C(\boldsymbol{\beta}) = (\boldsymbol{\tilde{y}} - \tilde{X}\boldsymbol{\beta})^T
|
||||
which has the solution
|
||||
|
||||
$\beta = (\tilde{X}^T\tilde{X} + \lambda I)^{-1}\tilde{X}^T\boldsymbol{\tilde{y}}$.
|
||||
where $\boldsymbol{\tilde{y}} = \boldsymbol{y} - y_{average}$
|
||||
where $\boldsymbol{\tilde{y}} = \boldsymbol{y} - \overline{\bm{y}}$
|
||||
and $\tilde{X}_{ij} = X_{ij} - \frac{1}{n}\sum_{k=1}^{n-1}X_{kj}$.
|
||||
|
||||
!split
|
||||
===== Code Examples =====
|
||||
|
||||
|
||||
Armed with this wisdom, we attempt first simply set the intercept eqault to _False_ in our implementation of Ridge regression for a vanilla data set.
|
||||
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
@@ -507,8 +491,6 @@ import matplotlib.pyplot as plt
|
||||
from sklearn.model_selection import train_test_split
|
||||
from sklearn import linear_model
|
||||
|
||||
def R2(y_data, y_model):
|
||||
return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
@@ -524,42 +506,29 @@ y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 20
|
||||
X = np.zeros((n,Maxpolydegree))
|
||||
We include explicitely the intercpt column
|
||||
X[:,0] = 1.0
|
||||
|
||||
for polydegree in range(1, Maxpolydegree):
|
||||
for degree in range(polydegree):
|
||||
X[:,degree] = x**degree
|
||||
for degree in range(Maxpolydegree):
|
||||
X[:,degree] = x**degree
|
||||
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
# matrix inversion to find beta
|
||||
OLSbeta = np.linalg.pinv(X_train.T @ X_train) @ X_train.T @ y_train
|
||||
print(OLSbeta)
|
||||
# and then make the prediction
|
||||
ytildeOLS = X_train @ OLSbeta
|
||||
print("Training MSE for OLS")
|
||||
print(MSE(y_train,ytildeOLS))
|
||||
ypredictOLS = X_test @ OLSbeta
|
||||
print("Test MSE OLS")
|
||||
print(MSE(y_test,ypredictOLS))
|
||||
|
||||
p = len(OLSbeta)
|
||||
p = Maxpolydegree
|
||||
I = np.eye(p,p)
|
||||
# Decide which values of lambda to use
|
||||
nlambdas = 4
|
||||
MSEOwnRidgePredict = np.zeros(nlambdas)
|
||||
MSEOwnRidgeTrain = np.zeros(nlambdas)
|
||||
MSERidgePredict = np.zeros(nlambdas)
|
||||
MSERidgeTrain = np.zeros(nlambdas)
|
||||
|
||||
lambdas = np.logspace(-4, 4, nlambdas)
|
||||
for i in range(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train.T @ X_train+lmb*I) @ X_train.T @ y_train
|
||||
# include lasso using Scikit-Learn
|
||||
# Note: we include the intercept
|
||||
# Note: we include the intercept column and no scaling
|
||||
RegRidge = linear_model.Ridge(lmb,fit_intercept=False)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
# and then make the prediction
|
||||
@@ -568,18 +537,14 @@ for i in range(nlambdas):
|
||||
ytildeRidge = RegRidge.predict(X_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSEOwnRidgeTrain[i] = MSE(y_train,ytildeOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
MSERidgeTrain[i] = MSE(y_train,ytildeRidge)
|
||||
print("Beta values for own Ridge implementation")
|
||||
print(OwnRidgeBeta)
|
||||
print("Beta values for Scikit-Learn Ridge implementation")
|
||||
print(RegRidge.coef_)
|
||||
# Now plot the results
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgeTrain, 'b', label = 'MSE Ridge train')
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'r', label = 'MSE Ridge Test')
|
||||
plt.plot(np.log10(lambdas), MSERidgeTrain, 'y', label = 'MSE Ridge train')
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, 'g', label = 'MSE Ridge Test')
|
||||
|
||||
plt.xlabel('log10(lambda)')
|
||||
@@ -589,8 +554,13 @@ plt.show()
|
||||
|
||||
!ec
|
||||
|
||||
The results here agree when we force _Scikit-Learn_'s Ridge function to include the first column in our design matrix.
|
||||
The problem however is that can easily lead to a larger mean-squared error!
|
||||
|
||||
Let us see how we can change this code by zero centering.
|
||||
|
||||
!split
|
||||
===== Taking out the mean =====
|
||||
!bc pycod
|
||||
import numpy as np
|
||||
import pandas as pd
|
||||
@@ -599,13 +569,9 @@ from sklearn.model_selection import train_test_split
|
||||
from sklearn import linear_model
|
||||
from sklearn.preprocessing import StandardScaler
|
||||
|
||||
def R2(y_data, y_model):
|
||||
return 1 - np.sum((y_data - y_model) ** 2) / np.sum((y_data - np.mean(y_data)) ** 2)
|
||||
def MSE(y_data,y_model):
|
||||
n = np.size(y_model)
|
||||
return np.sum((y_data-y_model)**2)/n
|
||||
|
||||
|
||||
# A seed just to ensure that the random numbers are the same for every run.
|
||||
# Useful for eventual debugging.
|
||||
np.random.seed(315)
|
||||
@@ -614,15 +580,12 @@ n = 100
|
||||
x = np.random.rand(n)
|
||||
y = np.exp(-x**2) + 1.5 * np.exp(-(x-2)**2)
|
||||
|
||||
Maxpolydegree = 5
|
||||
Maxpolydegree = 20
|
||||
X = np.zeros((n,Maxpolydegree-1))
|
||||
|
||||
for degree in range(1,Maxpolydegree): #No intercept column
|
||||
X[:,degree-1] = x**(degree)
|
||||
|
||||
|
||||
|
||||
|
||||
# We split the data in test and training data
|
||||
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
@@ -632,11 +595,13 @@ X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)
|
||||
|
||||
#For our own implementation, we will need to deal with the intercept by centering the design matrix and the target variable
|
||||
X_train_mean = np.mean(X_train,axis=0)
|
||||
X_train_scaled = X_train - X_train_mean #Center by removing mean from each feature
|
||||
#Center by removing mean from each feature
|
||||
X_train_scaled = X_train - X_train_mean
|
||||
X_test_scaled = X_test - X_train_mean
|
||||
|
||||
y_scaler = np.mean(y_train) #The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)
|
||||
y_train_scaled = y_train - y_scaler #Remove the intercept from the training data.
|
||||
#The model intercept (called y_scaler) is given by the mean of target variable (IF X is centered)
|
||||
#Remove the intercept from the training data.
|
||||
y_scaler = np.mean(y_train)
|
||||
y_train_scaled = y_train - y_scaler
|
||||
|
||||
|
||||
p = Maxpolydegree-1
|
||||
@@ -651,25 +616,20 @@ for i in range(nlambdas):
|
||||
lmb = lambdas[i]
|
||||
OwnRidgeBeta = np.linalg.pinv(X_train_scaled.T @ X_train_scaled+lmb*I) @ X_train_scaled.T @ (y_train_scaled)
|
||||
intercept_ = y_scaler - X_train_mean@OwnRidgeBeta #The intercept can be shifted so the model can predict on uncentered data
|
||||
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_ #Add intercept to prediction
|
||||
#Add intercept to prediction
|
||||
ypredictOwnRidge = X_test @ OwnRidgeBeta + intercept_
|
||||
#EQUIVALENT PREDICTION:
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler #Add intercept to prediction
|
||||
#Add intercept to prediction
|
||||
ypredictOwnRidge = X_test_scaled @ OwnRidgeBeta + y_scaler
|
||||
print("Values for own Ridge prediction")
|
||||
print(ypredictOwnRidge)
|
||||
|
||||
|
||||
|
||||
RegRidge = linear_model.Ridge(lmb)
|
||||
RegRidge.fit(X_train,y_train)
|
||||
ypredictRidge = RegRidge.predict(X_test)
|
||||
print("Values for SL Ridge prediction")
|
||||
print(ypredictRidge)
|
||||
|
||||
|
||||
MSEOwnRidgePredict[i] = MSE(y_test,ypredictOwnRidge)
|
||||
MSERidgePredict[i] = MSE(y_test,ypredictRidge)
|
||||
|
||||
print("Beta values for own Ridge implementation")
|
||||
print(OwnRidgeBeta) #Intercept is given by mean of target variable
|
||||
print("Beta values for Scikit-Learn Ridge implementation")
|
||||
@@ -679,20 +639,15 @@ for i in range(nlambdas):
|
||||
print('Intercept from Scikit-Learn Ridge implementation')
|
||||
print(RegRidge.intercept_)
|
||||
|
||||
|
||||
|
||||
# Now plot the results
|
||||
|
||||
plt.figure()
|
||||
plt.plot(np.log10(lambdas), MSEOwnRidgePredict, 'b--', label = 'MSE own Ridge Test')
|
||||
plt.plot(np.log10(lambdas), MSERidgePredict, 'g--', label = 'MSE SL Ridge Test')
|
||||
|
||||
plt.xlabel('log10(lambda)')
|
||||
plt.ylabel('MSE')
|
||||
plt.legend()
|
||||
plt.show()
|
||||
|
||||
|
||||
!ec
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user